diff --git a/Cargo.lock b/Cargo.lock index 0ba9c6e2c..ff301e7f1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -865,7 +865,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" dependencies = [ "digest 0.10.7", - "spin 0.10.0", + "spin 0.10.1", ] [[package]] @@ -1376,7 +1376,7 @@ dependencies = [ "futures-core", "futures-sink", "nanorand", - "spin 0.9.8", + "spin 0.9.9", ] [[package]] @@ -2585,9 +2585,9 @@ dependencies = [ [[package]] name = "mio" -version = "1.2.1" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "02bd0af71c67b473010cbbc60715ee815645a4dc942899111f494b4b737d6fda" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" dependencies = [ "libc", "log", @@ -2616,9 +2616,9 @@ dependencies = [ [[package]] name = "napi" -version = "3.10.4" +version = "3.10.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "59b7fbd5f12adbf51ddec954d4ef9cecb3542a9d53bce2fe0653696c4cd06d73" +checksum = "6826e5ddc15589b2d68c8ad5321c18e85d40488e93e32962f362e572669bccf6" dependencies = [ "bitflags 2.13.0", "ctor", @@ -3264,7 +3264,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.4.8" +version = "16.5.0" dependencies = [ "anyhow", "ast-grep-core", @@ -3333,7 +3333,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.4.8" +version = "16.5.0" dependencies = [ "async-trait", "libc", @@ -3345,7 +3345,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.4.8" +version = "16.5.0" dependencies = [ "anyhow", "arboard", @@ -3398,7 +3398,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.4.8" +version = "16.5.0" dependencies = [ "anyhow", "brush-builtins", @@ -3482,7 +3482,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.4.8" +version = "16.5.0" dependencies = [ "dashmap", "globset", @@ -4200,9 +4200,9 @@ dependencies = [ [[package]] name = "socket2" -version = "0.6.4" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52d1cfed4120b4d927bf7c0f86d2087a4a7d6027c906d9f9d525a80573b9be51" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", "windows-sys 0.61.2", @@ -4210,18 +4210,18 @@ dependencies = [ [[package]] name = "spin" -version = "0.9.8" +version = "0.9.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" +checksum = "3763264f6b73151db08c50ff20d7d8a0b8796e021cdea7ceedad07b80155fa0e" dependencies = [ "lock_api", ] [[package]] name = "spin" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" +checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" [[package]] name = "stable_deref_trait" @@ -4535,7 +4535,7 @@ dependencies = [ "toml_datetime", "toml_parser", "toml_writer", - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] @@ -4556,7 +4556,7 @@ dependencies = [ "indexmap", "toml_datetime", "toml_parser", - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] @@ -4565,7 +4565,7 @@ version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" dependencies = [ - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] @@ -6007,9 +6007,9 @@ checksum = "0bb6d972f580f8223cb7052d8580aea2b7061e368cf476de32ea9457b19459ed" [[package]] name = "uuid" -version = "1.23.4" +version = "1.23.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf80a72845275afea99e7f2b434723d3bc7e38470fcd1c7ed39a599c73319a53" +checksum = "ea5fab0d6c3c01ae70085a09cb03d4c7a1d6314e2b3e075392783396d724ca0a" dependencies = [ "js-sys", "wasm-bindgen", @@ -6607,9 +6607,9 @@ dependencies = [ [[package]] name = "winnow" -version = "1.0.3" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0592e1c9d151f854e6fd382574c3a0855250e1d9b2f99d9281c6e6391af352f1" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" dependencies = [ "memchr", ] @@ -6836,9 +6836,9 @@ dependencies = [ [[package]] name = "zmij" -version = "1.0.21" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" [[package]] name = "zune-core" diff --git a/Cargo.toml b/Cargo.toml index c603960b2..3057d0683 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.4.8" +version = "16.5.0" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index ed47a4f51..6092c721c 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.8", + "version": "16.5.0", "bin": { "omp": "src/cli.ts", }, @@ -136,16 +136,28 @@ "@types/react-dom": "catalog:", }, }, - "packages/harbor-manager": { - "name": "@oh-my-pi/harbor-manager", + "packages/hashline": { + "name": "@oh-my-pi/hashline", + "version": "16.5.0", + "dependencies": { + "diff": "catalog:", + "lru-cache": "catalog:", + }, + "devDependencies": { + "@types/bun": "catalog:", + }, + }, + "packages/metaharness": { + "name": "@oh-my-pi/pi-metaharness", "version": "0.0.1", "bin": { - "harbor-manager": "src/server.ts", + "metaharness": "src/server.ts", }, "dependencies": { "@oh-my-pi/hashline": "catalog:", "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "@oh-my-pi/typescript-edit-benchmark": "workspace:*", @@ -164,24 +176,12 @@ "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", "@types/react-dom": "^19.1.0", - "@vitejs/plugin-react": "^5.0.4", - "vite": "catalog:", - }, - }, - "packages/hashline": { - "name": "@oh-my-pi/hashline", - "version": "16.4.8", - "dependencies": { - "diff": "catalog:", - "lru-cache": "catalog:", - }, - "devDependencies": { - "@types/bun": "catalog:", + "react-refresh": "^0.18.0", }, }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.4.8", + "version": "16.5.0", "bin": { "mnemopi": "src/cli.ts", }, @@ -207,7 +207,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.4.8", + "version": "16.5.0", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -215,7 +215,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -228,7 +228,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.4.8", + "version": "16.5.0", "bin": { "omp-stats": "./src/index.ts", }, @@ -254,7 +254,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.4.8", + "version": "16.5.0", "bin": { "omp-swarm": "src/cli.ts", }, @@ -270,7 +270,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -308,7 +308,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.4.8", + "version": "16.5.0", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -321,7 +321,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.4.8", + "version": "16.5.0", "devDependencies": { "@types/bun": "catalog:", }, @@ -343,93 +343,94 @@ }, }, "patchedDependencies": { - "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", + "@ark/schema@0.56.2": "patches/@ark%2Fschema@0.56.2.patch", + "@agentclientprotocol/sdk@1.2.1": "patches/@agentclientprotocol%2Fsdk@1.2.1.patch", }, "overrides": { - "@ark/schema": "0.56.1", + "@ark/schema": "0.56.2", }, "catalog": { - "@agentclientprotocol/sdk": "0.25.0", + "@agentclientprotocol/sdk": "1.2.1", "@babel/generator": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.4.16", - "@bufbuild/protobuf": "^2.12.0", - "@bufbuild/protoc-gen-es": "^2.12.0", + "@biomejs/biome": "^2.5.3", + "@bufbuild/protobuf": "^2.12.1", + "@bufbuild/protoc-gen-es": "^2.12.1", "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", - "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.8", - "@oh-my-pi/omp-stats": "16.4.8", - "@oh-my-pi/pi-agent-core": "16.4.8", - "@oh-my-pi/pi-ai": "16.4.8", - "@oh-my-pi/pi-catalog": "16.4.8", - "@oh-my-pi/pi-coding-agent": "16.4.8", - "@oh-my-pi/pi-mnemopi": "16.4.8", - "@oh-my-pi/pi-natives": "16.4.8", - "@oh-my-pi/pi-tui": "16.4.8", - "@oh-my-pi/pi-utils": "16.4.8", - "@oh-my-pi/pi-wire": "16.4.8", - "@oh-my-pi/snapcompact": "16.4.8", + "@napi-rs/cli": "3.7.2", + "@oh-my-pi/hashline": "16.5.0", + "@oh-my-pi/omp-stats": "16.5.0", + "@oh-my-pi/pi-agent-core": "16.5.0", + "@oh-my-pi/pi-ai": "16.5.0", + "@oh-my-pi/pi-catalog": "16.5.0", + "@oh-my-pi/pi-coding-agent": "16.5.0", + "@oh-my-pi/pi-mnemopi": "16.5.0", + "@oh-my-pi/pi-natives": "16.5.0", + "@oh-my-pi/pi-tui": "16.5.0", + "@oh-my-pi/pi-utils": "16.5.0", + "@oh-my-pi/pi-wire": "16.5.0", + "@oh-my-pi/snapcompact": "16.5.0", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", - "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", + "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", "@opentelemetry/resources": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", "@opentelemetry/sdk-trace-node": "^2.7.1", - "@puppeteer/browsers": "^3.0.4", - "@tailwindcss/node": "^4.3.0", - "@tailwindcss/vite": "^4.3.0", + "@puppeteer/browsers": "^3.0.6", + "@tailwindcss/node": "^4.3.2", + "@tailwindcss/vite": "^4.3.2", "@types/babel__generator": "^7.27.0", "@types/babel__traverse": "^7.28.0", "@types/bun": "^1.3.14", "@types/react": "^19.2.17", "@types/react-dom": "^19.2.3", "@types/turndown": "5.0.6", - "@typescript/native-preview": "7.0.0-dev.20260609.1", + "@typescript/native-preview": "7.0.0-dev.20260707.2", "@xterm/headless": "^6.0.0", - "arktype": "2.2.2", + "arktype": "2.2.3", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", "diff": "^9.0.0", - "fast-xml-parser": "^5.9.0", + "fast-xml-parser": "^5.9.3", "fastembed": "2.1.0", "fflate": "0.8.3", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", "header-generator": "^2.1.82", - "linkedom": "^0.18.12", - "lint-staged": "^17.0.7", - "lru-cache": "11.5.1", - "lucide-react": "^1.17.0", + "linkedom": "^0.18.13", + "lint-staged": "^17.0.8", + "lru-cache": "11.5.2", + "lucide-react": "^1.24.0", "mammoth": "^1.12.0", - "marked": "^18.0.5", - "mupdf": "^1.27.0", + "marked": "^18.0.6", + "mupdf": "^1.28.0", "onnxruntime-node": "1.26.0", - "postcss": "^8.5.15", - "prettier": "^3.8.4", + "postcss": "^8.5.16", + "prettier": "^3.9.5", "puppeteer-core": "25.3.0", "react": "19.2.7", "react-chartjs-2": "^5.3.1", "react-dom": "19.2.7", "regexp-tree": "^0.1.27", - "solid-js": "^1.9.13", - "tailwindcss": "^4.3.0", + "solid-js": "^1.9.14", + "tailwindcss": "^4.3.2", "ts-morph": "^28.0.0", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", - "typescript": "^6.0.3", - "vite": "^8.0.16", + "typescript": "^7.0.2", + "vite": "^8.1.4", "vite-plugin-solid": "^2.11.12", "winston": "^3.19.0", "winston-daily-rotate-file": "^5.0.0", "zod": "^4", }, "packages": { - "@agentclientprotocol/sdk": ["@agentclientprotocol/sdk@0.25.0", "", { "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" } }, "sha512-wU1VgXNtMvdVotX49txc3WJUDV+/QbLpsgjMvFhlRmp37osdLbI7L7y+iwAlQATwfjLxcv1r1p3ZxZBcXlGhcQ=="], + "@agentclientprotocol/sdk": ["@agentclientprotocol/sdk@1.2.1", "", { "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" } }, "sha512-jwYUdOQR7tc+Zfch53VL4JJyUNK/46q03uUTYb+PjECsmnNl94XFXOfYLJ8RBpMNidXd1rpOAVgb0vqD98xImA=="], "@anush008/tokenizers": ["@anush008/tokenizers@0.0.0", "", { "optionalDependencies": { "@anush008/tokenizers-darwin-universal": "0.0.0", "@anush008/tokenizers-linux-x64-gnu": "0.0.0", "@anush008/tokenizers-win32-x64-msvc": "0.0.0" } }, "sha512-IQD9wkVReKAhsEAbDjh/0KrBGTEXelqZLpOBRDaIRvlzZ9sjmUP+gKbpvzyJnei2JHQiE8JAgj7YcNloINbGBw=="], @@ -439,9 +440,9 @@ "@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="], - "@ark/schema": ["@ark/schema@0.56.1", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-1Cf2g9nKD8K/3JGRu+gCCfYw5d4qR8YLLjDs5W5kpmaButCYWAPFUJqSXyBATPjglzCd4tIkp398iPYVs8MjRA=="], + "@ark/schema": ["@ark/schema@0.56.2", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-Qx4D2JFbBWpntiHZaTv7bGG4H/M2rigiknezKg/WVyDSaLdE4YCcWAOoFB7pjjDqHbbV2OqRfntm1nnXvwMexg=="], - "@ark/util": ["@ark/util@0.56.1", "", {}, "sha512-Tp1rTik3q5Z+jAeeDxr5JZpmVIw0miti1ykSEHyZv5Pw3TIJl2xbN6KTacOxITp0l3s9ytlrWd30Zvqcy5vzoQ=="], + "@ark/util": ["@ark/util@0.56.2", "", {}, "sha512-9kU2sUE38FZEGG7l3hamYMBieLYEJh2L1mrYD2eXpT+78EnQSV1bhjxJhnxGBMSTbtwpBSDNSK+K60WvaI/DTQ=="], "@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="], @@ -473,10 +474,6 @@ "@babel/plugin-syntax-jsx": ["@babel/plugin-syntax-jsx@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-TSu8+mHCoEaaCDEZ0I3+6mvTBYR4PCxQwf2z9/r5Tbztv6NaLR3B9thGTTxX2WGuGHJqRiAbKPeGTJ5XWXVg6A=="], - "@babel/plugin-transform-react-jsx-self": ["@babel/plugin-transform-react-jsx-self@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-TL0hMc9xzy86VD31nUiwzd5otRAcyEPcsegCxolO0PvcXuH1v0kECe/UIznYFihpkvU5wg/jk4v0TTEFfm53fw=="], - - "@babel/plugin-transform-react-jsx-source": ["@babel/plugin-transform-react-jsx-source@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-06IyK09H3wi4cGbhDBwp5gUGo0IKtnYa8tyTiephirPCK6fbobVGiXMMI5zLQ4aKEYP3wZ3ArU44o+8KMrSG/Q=="], - "@babel/template": ["@babel/template@7.29.7", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-puq+Gf35oI24FeN11LkoUQFqv9uwNeWpxXZi/Ji3rRIoKAzKnxRaZ+Gkj0vKS9ZCiTESfng1N9LyOyXvo+m+Gg=="], "@babel/traverse": ["@babel/traverse@7.29.7", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/generator": "^7.29.7", "@babel/helper-globals": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/template": "^7.29.7", "@babel/types": "^7.29.7", "debug": "^4.3.1" } }, "sha512-EhlfNQtZ+NK22w5BM61ciuiq1m58ed33Wr1Xan//ZRTy6hgjnwyCffRYwzsGXdASJSUJ1guZILsErh1eQcl+zw=="], @@ -631,7 +628,7 @@ "@mozilla/readability": ["@mozilla/readability@0.6.0", "", {}, "sha512-juG5VWh4qAivzTAeMzvY9xs9HY5rAcr2E4I7tiSSCokRFi7XIZCAu92ZkSTsIj1OPceCifL3cpfteP3pDT9/QQ=="], - "@napi-rs/cli": ["@napi-rs/cli@3.7.0", "", { "dependencies": { "@inquirer/prompts": "^8.0.0", "@napi-rs/cross-toolchain": "^1.0.3", "@napi-rs/wasm-tools": "^1.0.1", "@octokit/rest": "^22.0.1", "clipanion": "^4.0.0-rc.4", "colorette": "^2.0.20", "emnapi": "^1.10.0", "es-toolkit": "^1.41.0", "js-yaml": "^4.1.0", "obug": "^2.0.0", "semver": "^7.7.3", "typanion": "^3.14.0" }, "peerDependencies": { "@emnapi/runtime": "^1.7.1" }, "optionalPeers": ["@emnapi/runtime"], "bin": { "napi": "dist/cli.js", "napi-raw": "cli.mjs" } }, "sha512-3d3+rmxlOIV/G1zPWeX4PCxuYnhcCQM2BvY9rtimC8RO0dFR9gtYP+Grov+WoduZtfWRj5N1XvytWeRxxCk5zw=="], + "@napi-rs/cli": ["@napi-rs/cli@3.7.2", "", { "dependencies": { "@inquirer/prompts": "^8.5.2", "@napi-rs/cross-toolchain": "^1.0.3", "@napi-rs/wasm-tools": "^1.0.1", "@octokit/rest": "^22.0.1", "clipanion": "^4.0.0-rc.4", "colorette": "^2.0.20", "emnapi": "^1.11.1", "es-toolkit": "^1.47.0", "js-yaml": "^4.2.0", "obug": "^2.1.2", "semver": "^7.8.2", "typanion": "^3.14.0" }, "peerDependencies": { "@emnapi/runtime": "^1.7.1" }, "optionalPeers": ["@emnapi/runtime"], "bin": { "napi": "dist/cli.js", "napi-raw": "cli.mjs" } }, "sha512-shDW0Td/XZQpP04Yy+OsMt1ILMKGGkoLcy1zVAsSAK0fLfWm0Upgkmfs/NOV2ZhMQwkgpR3ZEdyHmTwgrUDQuA=="], "@napi-rs/cross-toolchain": ["@napi-rs/cross-toolchain@1.0.3", "", { "dependencies": { "@napi-rs/lzma": "^1.4.5", "@napi-rs/tar": "^1.1.0", "debug": "^4.4.1" }, "peerDependencies": { "@napi-rs/cross-toolchain-arm64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-x86_64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-x86_64": "^1.0.3" }, "optionalPeers": ["@napi-rs/cross-toolchain-arm64-target-aarch64", "@napi-rs/cross-toolchain-arm64-target-armv7", "@napi-rs/cross-toolchain-arm64-target-ppc64le", "@napi-rs/cross-toolchain-arm64-target-s390x", "@napi-rs/cross-toolchain-arm64-target-x86_64", "@napi-rs/cross-toolchain-x64-target-aarch64", "@napi-rs/cross-toolchain-x64-target-armv7", "@napi-rs/cross-toolchain-x64-target-ppc64le", "@napi-rs/cross-toolchain-x64-target-s390x", "@napi-rs/cross-toolchain-x64-target-x86_64"] }, "sha512-ENPfLe4937bsKVTDA6zdABx4pq9w0tHqRrJHyaGxgaPq03a2Bd1unD5XSKjXJjebsABJ+MjAv1A2OvCgK9yehg=="], @@ -763,8 +760,6 @@ "@oh-my-pi/collab-web": ["@oh-my-pi/collab-web@workspace:packages/collab-web"], - "@oh-my-pi/harbor-manager": ["@oh-my-pi/harbor-manager@workspace:packages/harbor-manager"], - "@oh-my-pi/hashline": ["@oh-my-pi/hashline@workspace:packages/hashline"], "@oh-my-pi/omp-stats": ["@oh-my-pi/omp-stats@workspace:packages/stats"], @@ -777,6 +772,8 @@ "@oh-my-pi/pi-coding-agent": ["@oh-my-pi/pi-coding-agent@workspace:packages/coding-agent"], + "@oh-my-pi/pi-metaharness": ["@oh-my-pi/pi-metaharness@workspace:packages/metaharness"], + "@oh-my-pi/pi-mnemopi": ["@oh-my-pi/pi-mnemopi@workspace:packages/mnemopi"], "@oh-my-pi/pi-natives": ["@oh-my-pi/pi-natives@workspace:packages/natives"], @@ -795,23 +792,23 @@ "@opentelemetry/api": ["@opentelemetry/api@1.9.1", "", {}, "sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q=="], - "@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-fmEWp5kXlGEc3i/lR698Hz41DfGyN4Tbe4g7L1AxSc7fF8Xeh/FQ9Quqpa9dVA413Q1Ad43QOLzU4JoXgbFPWw=="], + "@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.220.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CmVa4ImJ+ynfrPMNaAXHET6Bhb44SwzmfyVJFq9ni2jgXJR/l7C6gfVFddNmHP+ZOkP9cf4f9DBe68qVLTHc9w=="], "@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.9.0", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-OQ0vzvbZBiUhjqLnUaoNfYmP8553Crr3aggB4y0ZUi815mZ7idpdJXQmoKdeBKJelYttoBlLSSHubmyw3wvX4w=="], "@opentelemetry/core": ["@opentelemetry/core@2.9.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw=="], - "@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.218.0", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/otlp-exporter-base": "0.218.0", "@opentelemetry/otlp-transformer": "0.218.0", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-r1Msf8SNLRmwh9J6XQ5uh82D7CdDWMNHnPB7LAVHjzut0TkSeKc5KcIvr4SvHvfk/xwN5gxC+VLKQ1k0o8PSPw=="], + "@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-trace": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-voTAD8XgJxlK7zLkXh8EzMB09zrQr3tyY/BsnDTlDiQU/UdK58MZ63A3mUjdEDrxMjCVmBHU3WQJhRmQe+Dvzg=="], - "@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.218.0", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/otlp-transformer": "0.218.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-ZwqpkNL5W7RyGJPDZ9g06DvKp8KFTWPJPN12anpMQYSKpTSU0z3EIZuPq9vPGpS8siFyOqDYDAuCwlNO9FqgbA=="], + "@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-transformer": "0.220.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CXYo8UD5Mn9YbgebO2EL4wejtA+gxLmLiu6HCk2KH2BR7XhFN6/6p1UlCb23DYCjeYkndevLHuejCCN1yx4+OQ=="], - "@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-logs": "0.218.0", "@opentelemetry/sdk-metrics": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CFaKH87WAzjuJ4awowTTLzUvMfaRfiOFG5+qm5S5ncyalRtN4ecQ+YmuANJSCrVPuvZFEkUgKhBPBndxi3rHsQ=="], + "@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.220.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.220.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-logs": "0.220.0", "@opentelemetry/sdk-metrics": "2.9.0", "@opentelemetry/sdk-trace": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-lXGrv7KXZ0gNH9SVNUaa6vv6phVYGvJxfXAlMbzbakiXru75f5MZl8Z7oqiMMQD77riVHJCFlQvbZs/VVN2/4A=="], "@opentelemetry/resources": ["@opentelemetry/resources@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-jyA5MBLQ+Dkl3+JsZkUoUvL7yHvU64kLsvpXKarWm6347Sl1t1bXFTFykUePNpT5WH5pm9a2Qtt03iIYQhZ1Fg=="], - "@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-QvnNdugatFTVCJXH0Mcu7GOOJSylA9j127kIezOE4YwTI4YbowRons2K4WZTv5FMS8T4q9P0NdaRHdkSmeAIag=="], + "@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.220.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.220.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-WywcTkQtv2iNmt+6y5Kcd4rzvx9bLVsBa2Nwcmg01IUaBTkTow3W4d9KE5vNBpEDtb9tp21WcRBY/lANRrApYA=="], - "@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-MpDJdkiFDs3Pm1RHO3KByuZbuBdJEXEAkiC0+yJdsZGVCdf1RpHR6n+LHDcS7ffmfrt5kVCzJSCfm4z2C7v0uQ=="], + "@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-Xx8RGS4H5XEBl01WuCreMIpiah9cCXMbSkeuIePPdD2cUpq/vUzYmj8E/MK1OsbOc93FuAD4jfn2WOacKwLn7Q=="], "@opentelemetry/sdk-trace": ["@opentelemetry/sdk-trace@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-sGA19HvtrrSKYsseHphluH6j3p6Xa3fqc7c7y8f/7mYWejc1lyDFcpSdD1kYa50HCLUeEo4zA5bW0pniaPszuw=="], @@ -819,7 +816,7 @@ "@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.9.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.9.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/sdk-trace-base": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-ec9a7ps37huy5itYk0MalaZdSLlM6AXWp/FhtEjgMpp5leEGojBDvAl/UWttQnkMZOvFHKzRESn8TD3yKTF5nQ=="], - "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.42.0", "", {}, "sha512-icc5xCzndZfhuJMy5oqk5AvloWquR7jtae74qzpkKkhGp8BivK+oCcEXgGnjCdTfp8hA44l+w8gE8yYJbocJJw=="], + "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.43.0", "", {}, "sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg=="], "@oxc-project/types": ["@oxc-project/types@0.139.0", "", {}, "sha512-r9gHphtCs+1M7J0pw6Sn/hh/Wpa/iQrOOkrNAlVLF/gHq+/CJmHIWKKUUhdWjcD6CIa8idarspCsASiXCXvFUw=="], @@ -873,7 +870,7 @@ "@rolldown/binding-win32-x64-msvc": ["@rolldown/binding-win32-x64-msvc@1.1.5", "", { "os": "win32", "cpu": "x64" }, "sha512-tTZuDBPw85tEN5PQi1pnEBzDy0Z49HtScLAbD5t6hyeU92A95pRWaSMw1GZZi/RwgSgUIl0xrSlXIT/9QzvYSA=="], - "@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.0-rc.3", "", {}, "sha512-eybk3TjzzzV97Dlj5c+XrBFW57eTNhzod66y9HrBlzJ6NsCrWCp/2kaPS3K9wJmurBC0Tdw4yPjXKZqlznim3Q=="], + "@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="], "@sindresorhus/is": ["@sindresorhus/is@4.6.0", "", {}, "sha512-t09vSN3MdfsyCHoFcTRCH/iUtG7OJ0CsjzB8cjAmKc/va/kIgeDI/TxsigdncE/4be734m0cvIYwNaV4i2XqAw=="], @@ -941,26 +938,64 @@ "@types/turndown": ["@types/turndown@5.0.6", "", {}, "sha512-ru00MoyeeouE5BX4gRL+6m/BsDfbRayOskWqUvh7CLGW+UXxHQItqALa38kKnOiZPqJrtzJUgAC2+F0rL1S4Pg=="], - "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260609.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260609.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260609.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-1HOuH/u/451O3hx4Z9fesNqarpeit6UfkgwK96sCVWi5p69F0N3v+6bI969lLIjF7K9dbYQNiWUaZ6Wik87iKg=="], + "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260707.2", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260707.2", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260707.2" }, "bin": { "tsgo": "bin/tsgo" } }, "sha512-oUGp+Rep/hqMhPunyinsALUwSlzHINSxitifPiSaeqoKOKD2OlR9NE3TaPqwsl4NlGslsOSUXI1JotWQzpYCPg=="], - "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260609.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-Yf/zHEadP/yUiWUdM/mZVfEVFJuGMf6nhRSFif0vp+FwtfGU4jmlpNF7BTJJdOHrrcWkwEJKzAoMCtEtyxhuyQ=="], + "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260707.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-wny2pgKjGbiZtnOIHVa3tXC1UfDqxNEFzyPGmiqybedG8hipG2Nfp0l5UxbaKCjkLacUpH/W5bP2hBOMVhCOzg=="], - "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260609.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-z4dYWI57CPHs0wV/FWFth8fWmqYH7iOm7THOfZ5Fv0jo/SWK6kE1kEUIqIAExqo7ueRNqSrCw0I8U1J4TJszAw=="], + "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260707.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-Afc7M5zOwo+GpfcYwz5Z8HMB2tPVsui7nNIqEuuFB73MPdVqNn/Wmpe4tP4MRri0AtJnJknoHBaTJ/VDAp/Jhw=="], - "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "arm" }, "sha512-mEtN8BbAgVtBu/5MVomYquXNvgok2C0KG6V0D4SV1jfBJNtlcqbp0WuIqT0bnM9DA4TgzcHvnFMpwGSK/dqI5A=="], + "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "arm" }, "sha512-hJm/UOqZTr9FHmR7uNm8VGX4oKtfWk0Jem0zPeJFNC8ckGUfSBueyiEYMZB+XmRc1aG4x1E46y3CplP4CLHvGQ=="], - "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-OxNVWH9IhrMAzNlDyDit1dPO64GFIDPOUKoruIkJ9A1ZEONfIHXG5f+V3si9jtuNmuomiz9FjpbzOqLsgaxt+w=="], + "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-iITBa2WjjTI5N9t5l7Z4KoOSI+2zBlhbvFzsD/f8qX8QoKjz/Y4DPyBDgezYi8nkqjjksbgSOJ3/ykzhwrB9cg=="], - "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "x64" }, "sha512-KO8WO1gBIC09T3255RlTY42TGu8en5mEkLPQu2wkMn+dX2T8KYL64zXrCeLeUWa0NvmVdJUeyWu3pFOn3zKemw=="], + "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "x64" }, "sha512-du0dzi6y97Po5vDNdPJTyyijHCpaS22JLRnKZEJXBDaO9gCIymOv/5QQokFRuOlQm0bWl3i9PF4OVdGP6uAOQA=="], - "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260609.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-+8q19LWjnMKK6SF3PLeMEalbfWDYWHs0AU8kSFCBCke/RLoDG4FjQzVtLgUo+KWhsmZMosiEyqEnZmSlED2tIQ=="], + "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260707.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-SsAwfhyHJ1akgBc+99z4+hwdbHsdWaKB8EwCNIMA6JfSLMeUjffrYvxu+vfMyxVtOVOz7RrRXRoiDiu4a2sCtg=="], - "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260609.1", "", { "os": "win32", "cpu": "x64" }, "sha512-qNPcss+6yRoNFfFIKQbPwJWYxDfOZwyL8JBJh4J+yMLOad/+/AOjsO4EtZsIpv5PMCjpnD75coBoDkw+5NkItw=="], + "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260707.2", "", { "os": "win32", "cpu": "x64" }, "sha512-DL4u27stv0fo71sVhOzHSwE+YMZsbBijVI+kg5dLDLilSH79WFTJ8RSQ46vJrCMt+Gjlv/JOZP1PuLJDfioYeQ=="], + + "@typescript/typescript-aix-ppc64": ["@typescript/typescript-aix-ppc64@7.0.2", "", { "os": "aix", "cpu": "ppc64" }, "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ=="], + + "@typescript/typescript-darwin-arm64": ["@typescript/typescript-darwin-arm64@7.0.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA=="], + + "@typescript/typescript-darwin-x64": ["@typescript/typescript-darwin-x64@7.0.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA=="], + + "@typescript/typescript-freebsd-arm64": ["@typescript/typescript-freebsd-arm64@7.0.2", "", { "os": "freebsd", "cpu": "arm64" }, "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ=="], + + "@typescript/typescript-freebsd-x64": ["@typescript/typescript-freebsd-x64@7.0.2", "", { "os": "freebsd", "cpu": "x64" }, "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw=="], + + "@typescript/typescript-linux-arm": ["@typescript/typescript-linux-arm@7.0.2", "", { "os": "linux", "cpu": "arm" }, "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ=="], + + "@typescript/typescript-linux-arm64": ["@typescript/typescript-linux-arm64@7.0.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ=="], + + "@typescript/typescript-linux-loong64": ["@typescript/typescript-linux-loong64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ=="], + + "@typescript/typescript-linux-mips64el": ["@typescript/typescript-linux-mips64el@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA=="], + + "@typescript/typescript-linux-ppc64": ["@typescript/typescript-linux-ppc64@7.0.2", "", { "os": "linux", "cpu": "ppc64" }, "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA=="], + + "@typescript/typescript-linux-riscv64": ["@typescript/typescript-linux-riscv64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ=="], + + "@typescript/typescript-linux-s390x": ["@typescript/typescript-linux-s390x@7.0.2", "", { "os": "linux", "cpu": "s390x" }, "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw=="], + + "@typescript/typescript-linux-x64": ["@typescript/typescript-linux-x64@7.0.2", "", { "os": "linux", "cpu": "x64" }, "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A=="], + + "@typescript/typescript-netbsd-arm64": ["@typescript/typescript-netbsd-arm64@7.0.2", "", { "os": "none", "cpu": "arm64" }, "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA=="], + + "@typescript/typescript-netbsd-x64": ["@typescript/typescript-netbsd-x64@7.0.2", "", { "os": "none", "cpu": "x64" }, "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA=="], + + "@typescript/typescript-openbsd-arm64": ["@typescript/typescript-openbsd-arm64@7.0.2", "", { "os": "openbsd", "cpu": "arm64" }, "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ=="], + + "@typescript/typescript-openbsd-x64": ["@typescript/typescript-openbsd-x64@7.0.2", "", { "os": "openbsd", "cpu": "x64" }, "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg=="], + + "@typescript/typescript-sunos-x64": ["@typescript/typescript-sunos-x64@7.0.2", "", { "os": "sunos", "cpu": "x64" }, "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g=="], + + "@typescript/typescript-win32-arm64": ["@typescript/typescript-win32-arm64@7.0.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ=="], + + "@typescript/typescript-win32-x64": ["@typescript/typescript-win32-x64@7.0.2", "", { "os": "win32", "cpu": "x64" }, "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g=="], "@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, "sha512-PJFXFS4ZJKiJ9Qiuix6Dz/OwEIqHD7Dme1UwZhTK11vR+5dqW2ACbdndWQexBzCx+CPuMe5WBYQWCsFyGlQLlQ=="], - "@vitejs/plugin-react": ["@vitejs/plugin-react@5.2.0", "", { "dependencies": { "@babel/core": "^7.29.0", "@babel/plugin-transform-react-jsx-self": "^7.27.1", "@babel/plugin-transform-react-jsx-source": "^7.27.1", "@rolldown/pluginutils": "1.0.0-rc.3", "@types/babel__core": "^7.20.5", "react-refresh": "^0.18.0" }, "peerDependencies": { "vite": "^4.2.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0" } }, "sha512-YmKkfhOAi3wsB1PhJq5Scj3GXMn3WvtQ/JC0xoopuHoXSdmtdStOpFrYaT1kie2YgFBcIe64ROzMYRjCrYOdYw=="], - "@xmldom/xmldom": ["@xmldom/xmldom@0.8.13", "", {}, "sha512-KRYzxepc14G/CEpEGc3Yn+JKaAeT63smlDr+vjB8jRfgTBBI9wRj/nkQEO+ucV8p8I9bfKLWp37uHgFrbntPvw=="], "@xterm/headless": ["@xterm/headless@6.0.0", "", {}, "sha512-5Yj1QINYCyzrZtf8OFIHi47iQtI+0qYFPHmouEfG8dHNxbZ9Tb9YGSuLcsEwj9Z+OL75GJqPyJbyoFer80a2Hw=="], @@ -977,9 +1012,9 @@ "argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="], - "arkregex": ["arkregex@0.0.7", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-O/Ltrn9EUSn3ui0KVzfyrWGDUsHlzKxDVBtpQxL/6JmLRMAZAebfSNf/A/J5Ny5S6QIwrXX+RfXsu888HMs35A=="], + "arkregex": ["arkregex@0.0.8", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-PJcx6G1kQTgLKPUbeYlYecDRaKq15AMSGVajlKFYWlPeJRQL+j3dKE6tyMs40HZ99djS1l9Vhl3ezAHy9JBIqQ=="], - "arktype": ["arktype@2.2.2", "", { "dependencies": { "@ark/schema": "0.56.1", "@ark/util": "0.56.1", "arkregex": "0.0.7" } }, "sha512-YYf1xhL2dh5aPZFlsY0RAsxv5HZqfLGLptH2ZP3JidTmsGRW8VOymhPjjMTkerL12vR2YtX0SK4c1mATtae8SA=="], + "arktype": ["arktype@2.2.3", "", { "dependencies": { "@ark/schema": "0.56.2", "@ark/util": "0.56.2", "arkregex": "0.0.8" } }, "sha512-7W+0RLTUNJiBFIIZXwOQxSR8Z273IAd6IvqBeG9+gHnQKFsIx2C0iOtGTmMrPnlX4qLXyc5+ll7A0BIj9WrbTg=="], "async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="], @@ -1279,7 +1314,7 @@ "lop": ["lop@0.4.2", "", { "dependencies": { "duck": "^0.1.12", "option": "~0.2.1", "underscore": "^1.13.1" } }, "sha512-RefILVDQ4DKoRZsJ4Pj22TxE3omDO47yFpkIBoDKzkqPRISs5U1cnAdg/5583YPkWPaLIYHOKRMQSvjFsO26cw=="], - "lru-cache": ["lru-cache@11.5.1", "", {}, "sha512-RPimw/7aMdv2oqRrxKwvZXcPfwBrn/JZ2xYcY9Hus/6LaS3VOAKVWKWgNLCFSiOm1ESXinjsDlidVU7JlnCN2A=="], + "lru-cache": ["lru-cache@11.5.2", "", {}, "sha512-4pfM1Ff0x50o0tQwb5ucw/RzNyD0/YJME6IVcStalZuMWxdt3sR3huStTtxz4PUmvZfRguvDejasvQ2kifR11g=="], "lucide-react": ["lucide-react@1.24.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, "sha512-YT6mBD8lGKkg4nM39enlm94/sfJIiW0YKUT60fBy4YK8tai31ylg1VhGNWxkpSKHo9UagfnZqwIff3HTDQwXeA=="], @@ -1495,7 +1530,7 @@ "typed-query-selector": ["typed-query-selector@2.12.2", "", {}, "sha512-EOPFbyIub4ngnEdqi2yOcNeDLaX/0jcE1JoAXQDDMIthap7FoN795lc/SHfIq2d416VufXpM8z/lD+WRm2gfOQ=="], - "typescript": ["typescript@6.0.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw=="], + "typescript": ["typescript@7.0.2", "", { "optionalDependencies": { "@typescript/typescript-aix-ppc64": "7.0.2", "@typescript/typescript-darwin-arm64": "7.0.2", "@typescript/typescript-darwin-x64": "7.0.2", "@typescript/typescript-freebsd-arm64": "7.0.2", "@typescript/typescript-freebsd-x64": "7.0.2", "@typescript/typescript-linux-arm": "7.0.2", "@typescript/typescript-linux-arm64": "7.0.2", "@typescript/typescript-linux-loong64": "7.0.2", "@typescript/typescript-linux-mips64el": "7.0.2", "@typescript/typescript-linux-ppc64": "7.0.2", "@typescript/typescript-linux-riscv64": "7.0.2", "@typescript/typescript-linux-s390x": "7.0.2", "@typescript/typescript-linux-x64": "7.0.2", "@typescript/typescript-netbsd-arm64": "7.0.2", "@typescript/typescript-netbsd-x64": "7.0.2", "@typescript/typescript-openbsd-arm64": "7.0.2", "@typescript/typescript-openbsd-x64": "7.0.2", "@typescript/typescript-sunos-x64": "7.0.2", "@typescript/typescript-win32-arm64": "7.0.2", "@typescript/typescript-win32-x64": "7.0.2" }, "bin": { "tsc": "bin/tsc" } }, "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA=="], "uglify-js": ["uglify-js@3.19.3", "", { "bin": { "uglifyjs": "bin/uglifyjs" } }, "sha512-v3Xu+yuwBXisp6QYTcH4UbH+xYJXqnq2m/LtQVWKWzYc1iehYnLixoQDN9FH6/j9/oybfd6W9Ghwkl8+UMKTKQ=="], @@ -1561,28 +1596,6 @@ "@isaacs/fs-minipass/minipass": ["minipass@7.1.3", "", {}, "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A=="], - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="], - - "@opentelemetry/otlp-exporter-base/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="], - - "@opentelemetry/sdk-logs/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/sdk-logs/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/sdk-metrics/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/sdk-metrics/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - "@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], @@ -1639,8 +1652,6 @@ "roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="], - "rolldown/@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="], - "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], "wrap-ansi/string-width": ["string-width@8.2.2", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-GaPUh5gfdrYzqeVNZvUfT23vYYxXzKYidUcnMtJg/3rxRV63EFZy3k6xfKlmfeJD0176lnUV/Usr3XcwSvFzpg=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 2ef74efb6..c878aed6d 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_4_8")] +#[napi(js_name = "__piNativesV16_5_0")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/docs/extensions.md b/docs/extensions.md index 7702c1767..786f66284 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -167,7 +167,7 @@ Handlers and tool `execute` receive `ctx` with: - `list()` — authenticated models available this session. - `current()` — the live session model (read lazily, so it reflects `/model` switches). -- `resolve(spec)` — a model string (`provider/id`, bare id) or role alias (`pi/slow`, a configured role) → `Model`, honoring the same settings-backed aliases and match preferences as `--model`. Returns `undefined` when nothing matches. +- `resolve(spec)` — a model string (`provider/id`, bare id) or role alias (`@slow`, a configured role) → `Model`, honoring the same settings-backed aliases and match preferences as `--model`. Returns `undefined` when nothing matches. - `family(model)` — an opaque lineage token for "same family?" checks (Claude point releases share a token; Claude and GPT differ). Compare it; don't persist it (the vocabulary tracks new releases). ```ts diff --git a/docs/local-models.md b/docs/local-models.md index 2a5a494d9..41de745f1 100644 --- a/docs/local-models.md +++ b/docs/local-models.md @@ -75,7 +75,7 @@ they opt in. | flan-t5-small | Rejected — just echoes the input | **Shipped local options**: `lfm2-350m`, `qwen3-0.6b`, `gemma-270m`, `qwen2.5-0.5b`, `lfm2-700m`. -**Default**: `online` (pi/smol). +**Default**: `online` (@smol). ## Task 2: Mnemopi memory (`providers.memoryModel`) diff --git a/docs/models.md b/docs/models.md index b6d3cc70d..cd4663b4d 100644 --- a/docs/models.md +++ b/docs/models.md @@ -446,9 +446,9 @@ Supported model roles: - `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor` -The `tiny` role overrides the online model used for lightweight background tasks (session titles, memory, `auto`-thinking difficulty classification, unexpected-stop detection); when unset, these fall back to `pi/smol`. Pick one in `/models`. +The `tiny` role overrides the online model used for lightweight background tasks (session titles, memory, `auto`-thinking difficulty classification, unexpected-stop detection); when unset, these fall back to `@smol`. Pick one in `/models`. -Role aliases like `pi/smol` expand through `settings.modelRoles`. Each role value can also append a thinking selector such as `:minimal`, `:low`, `:medium`, or `:high`. +Role aliases like `@smol` expand through `settings.modelRoles`; `*` selects `@default`. Quote `@` aliases in YAML values (`fable: "@slow"`). Each role value can also append a thinking selector such as `:minimal`, `:low`, `:medium`, or `:high`. If a role points at another role, the target model still inherits normally and any explicit suffix on the referring role wins for that role-specific use. diff --git a/docs/settings.md b/docs/settings.md index e56b620de..76be155f6 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -308,7 +308,7 @@ enabledModels: | Key | Type | Default | Notes | |---|---|---|---| -| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor`. The `tiny` role overrides the online model for lightweight background tasks (titles, memory, auto-thinking, unexpected-stop), else `pi/smol`. Per-role env/flags exist only for `--model`/`--smol`/`--slow`/`--plan`; configure the advisor with `modelRoles.advisor`. | +| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor`. The `tiny` role overrides the online model for lightweight background tasks (titles, memory, auto-thinking, unexpected-stop), else `@smol`. Per-role env/flags exist only for `--model`/`--smol`/`--slow`/`--plan`; configure the advisor with `modelRoles.advisor`. | | `modelTags` | record | `{}` | Custom role/tag metadata; can introduce additional roles. | | `modelProviderOrder` | array | `[]` | Preferred provider order when a model id is ambiguous. | | `cycleOrder` | array | `["smol","default","slow"]` | Roles cycled by the model switcher. | diff --git a/docs/tools/eval.md b/docs/tools/eval.md index be4dda38a..e41a97ee2 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -179,9 +179,9 @@ Both runtimes expose `completion()` — a single stateless completion against a - JS: `await completion(prompt, { model?, system?, schema? })` - Python: `completion(prompt, *, model="default", system=None, schema=None)` - `model` selects a tier (default `"default"`): - - `"smol"` → `pi/smol` role (fast / cheap) - - `"default"` → the session's active model, falling back to the `pi/default` role - - `"slow"` → `pi/slow` role; requests high reasoning effort only on reasoning-capable models + - `"smol"` → `@smol` role (fast / cheap) + - `"default"` → the session's active model, falling back to the `@default` role + - `"slow"` → `@slow` role; requests high reasoning effort only on reasoning-capable models - `system` (optional) supplies a system prompt. - `schema` (optional) is a plain JSON-Schema object. When present, the model is forced to call a single synthetic `respond` tool with that schema (loose, non-strict), and the helper returns the parsed object. When absent, the helper returns the completion string. - Errors surface as exceptions: unresolved tier, missing API key, an `error`/`aborted` stop reason, or empty output each raise. diff --git a/docs/tools/inspect_image.md b/docs/tools/inspect_image.md index aaea1fb08..36454eb70 100644 --- a/docs/tools/inspect_image.md +++ b/docs/tools/inspect_image.md @@ -40,7 +40,7 @@ TUI rendering adds presentation-only truncation from `packages/coding-agent/src/ ## Flow 1. `InspectImageTool.execute(...)` rejects immediately if `images.blockImages` is enabled in session settings. 2. It reads `session.modelRegistry`; missing registry, empty registry, missing API key, or unresolved model each raise `ToolError` from `packages/coding-agent/src/tools/inspect-image.ts`. -3. Model selection tries, in order, `pi/vision`, `pi/default`, the active model string from the session, then `availableModels[0]`. `expandRoleAlias(...)` and `resolveModelFromString(...)` handle each lookup. +3. Model selection tries, in order, `@vision`, `@default`, the active model string from the session, then `availableModels[0]`. `expandRoleAlias(...)` and `resolveModelFromString(...)` handle each lookup. 4. The chosen model must advertise `input.includes("image")`; otherwise execution fails before reading the file. 5. `loadImageInput(...)` in `packages/coding-agent/src/utils/image-loading.ts` resolves the path with `resolveReadPath(...)`, detects MIME type with `readImageMetadata(...)`, and rejects files larger than `MAX_IMAGE_INPUT_BYTES` (`20 * 1024 * 1024`, 20 MiB) using `ImageInputTooLargeError`. 6. `readImageMetadata(...)` in `packages/utils/src/mime.ts` inspects file headers only. Supported detected MIME types are `image/png`, `image/jpeg`, `image/gif`, and `image/webp`. diff --git a/package.json b/package.json index 7e88c3232..f1a3d26e2 100644 --- a/package.json +++ b/package.json @@ -5,8 +5,9 @@ "type": "module", "packageManager": "bun@1.3.14", "patchedDependencies": { - "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", - "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch" + "@ark/schema@0.56.2": "patches/@ark%2Fschema@0.56.2.patch", + "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", + "@agentclientprotocol/sdk@1.2.1": "patches/@agentclientprotocol%2Fsdk@1.2.1.patch" }, "workspaces": { "packages": [ @@ -14,79 +15,79 @@ "python/robomp/web" ], "catalog": { - "@agentclientprotocol/sdk": "0.25.0", + "@agentclientprotocol/sdk": "1.2.1", "@babel/generator": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.4.16", - "@bufbuild/protobuf": "^2.12.0", - "@bufbuild/protoc-gen-es": "^2.12.0", + "@biomejs/biome": "^2.5.3", + "@bufbuild/protobuf": "^2.12.1", + "@bufbuild/protoc-gen-es": "^2.12.1", "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", - "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.4.8", - "@oh-my-pi/omp-stats": "16.4.8", - "@oh-my-pi/pi-agent-core": "16.4.8", - "@oh-my-pi/pi-ai": "16.4.8", - "@oh-my-pi/pi-catalog": "16.4.8", - "@oh-my-pi/pi-coding-agent": "16.4.8", - "@oh-my-pi/pi-mnemopi": "16.4.8", - "@oh-my-pi/pi-natives": "16.4.8", - "@oh-my-pi/pi-tui": "16.4.8", - "@oh-my-pi/pi-utils": "16.4.8", - "@oh-my-pi/pi-wire": "16.4.8", - "@oh-my-pi/snapcompact": "16.4.8", + "@napi-rs/cli": "3.7.2", + "@oh-my-pi/hashline": "16.5.0", + "@oh-my-pi/omp-stats": "16.5.0", + "@oh-my-pi/pi-agent-core": "16.5.0", + "@oh-my-pi/pi-ai": "16.5.0", + "@oh-my-pi/pi-catalog": "16.5.0", + "@oh-my-pi/pi-coding-agent": "16.5.0", + "@oh-my-pi/pi-mnemopi": "16.5.0", + "@oh-my-pi/pi-natives": "16.5.0", + "@oh-my-pi/pi-tui": "16.5.0", + "@oh-my-pi/pi-utils": "16.5.0", + "@oh-my-pi/pi-wire": "16.5.0", + "@oh-my-pi/snapcompact": "16.5.0", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", - "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", + "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", "@opentelemetry/resources": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", "@opentelemetry/sdk-trace-node": "^2.7.1", - "@puppeteer/browsers": "^3.0.4", - "@tailwindcss/node": "^4.3.0", - "@tailwindcss/vite": "^4.3.0", + "@puppeteer/browsers": "^3.0.6", + "@tailwindcss/node": "^4.3.2", + "@tailwindcss/vite": "^4.3.2", "@types/babel__generator": "^7.27.0", "@types/babel__traverse": "^7.28.0", "@types/bun": "^1.3.14", "@types/react": "^19.2.17", "@types/react-dom": "^19.2.3", "@types/turndown": "5.0.6", - "@typescript/native-preview": "7.0.0-dev.20260609.1", + "@typescript/native-preview": "7.0.0-dev.20260707.2", "@xterm/headless": "^6.0.0", - "arktype": "2.2.2", + "arktype": "2.2.3", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", "diff": "^9.0.0", "fflate": "0.8.3", "fastembed": "2.1.0", - "fast-xml-parser": "^5.9.0", + "fast-xml-parser": "^5.9.3", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", "header-generator": "^2.1.82", - "linkedom": "^0.18.12", - "lint-staged": "^17.0.7", - "lru-cache": "11.5.1", - "lucide-react": "^1.17.0", + "linkedom": "^0.18.13", + "lint-staged": "^17.0.8", + "lru-cache": "11.5.2", + "lucide-react": "^1.24.0", "mammoth": "^1.12.0", - "marked": "^18.0.5", - "mupdf": "^1.27.0", + "marked": "^18.0.6", + "mupdf": "^1.28.0", "onnxruntime-node": "1.26.0", - "postcss": "^8.5.15", - "prettier": "^3.8.4", + "postcss": "^8.5.16", + "prettier": "^3.9.5", "puppeteer-core": "25.3.0", "react": "19.2.7", "react-chartjs-2": "^5.3.1", "react-dom": "19.2.7", "regexp-tree": "^0.1.27", - "solid-js": "^1.9.13", - "tailwindcss": "^4.3.0", + "solid-js": "^1.9.14", + "tailwindcss": "^4.3.2", "ts-morph": "^28.0.0", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", - "typescript": "^6.0.3", - "vite": "^8.0.16", + "typescript": "^7.0.2", + "vite": "^8.1.4", "vite-plugin-solid": "^2.11.12", "winston": "^3.19.0", "winston-daily-rotate-file": "^5.0.0", @@ -94,7 +95,7 @@ } }, "overrides": { - "@ark/schema": "0.56.1" + "@ark/schema": "0.56.2" }, "scripts": { "setup": "bun install && bun run build:native && bun --cwd=packages/coding-agent link && sh scripts/link-omp.sh", @@ -105,7 +106,7 @@ "collab:relay": "bun --cwd=packages/collab-web run relay", "collab:mock-host": "bun --cwd=packages/collab-web run mock-host", "collab:web:build": "bun --cwd=packages/collab-web run build", - "hmgr": "bun --cwd=packages/harbor-manager run dev", + "meta": "bun --cwd=packages/metaharness run dev", "claude:trace": "bun scripts/claude-trace.ts", "build": "bun run --workspaces --if-present build", "build:native": "bun --cwd=packages/natives run build", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index ba7b65aec..b5a968315 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,14 +2,12 @@ ## [Unreleased] +## [16.5.0] - 2026-07-13 + ### Added -- Added automated image-dropping rescue tier to compaction dead-end recovery -- Added visual warnings to the session timeline when compaction fails to free sufficient space - -### Changed - -- Improved compaction dead-end notifications with specific recovery instructions +- Added an automated image-dropping rescue tier to compaction dead-end recovery. +- Added visual warnings and detailed recovery instructions to the session timeline when compaction fails to free sufficient space. ## [16.4.5] - 2026-07-11 diff --git a/packages/agent/package.json b/packages/agent/package.json index ef3ea4a70..f0b8f0ab7 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.4.8", + "version": "16.5.0", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 927ed24cb..d3c2273a8 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,9 +2,23 @@ ## [Unreleased] +## [16.5.0] - 2026-07-13 + +### Added + +- Added diagnostic response headers to auth-gateway inference endpoints, including request IDs (x-request-id/request-id), LiteLLM model metadata (x-litellm-model-id/x-litellm-model-api-base), and performance/cost metrics (x-litellm-response-cost, x-litellm-response-duration-ms, openai-processing-ms) on non-streaming responses. + +### Changed + +- Updated Google and Google Vertex providers to always use streamGenerateContent requests. + ### Fixed -- Fixed empty provider responses (e.g. "Cloud Code Assist API returned an empty response") being classified as non-retryable: `ProviderResponseError` with kind `empty-body` now carries the transient flag, so session retry and configured model-fallback chains engage instead of hard-failing the turn +- Fixed empty provider responses (such as from Cloud Code Assist API) being classified as non-retryable, allowing session retries and model-fallback chains to engage instead of failing the turn. + +### Removed + +- Removed automatic /interactions chaining for follow-up turns in Google provider calls, along with the useInteractionsApi, storeInteraction, and previousInteractionId stream options. ## [16.4.6] - 2026-07-12 diff --git a/packages/ai/package.json b/packages/ai/package.json index 2ebedffb1..b7732aa37 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.4.8", + "version": "16.5.0", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/src/auth-gateway/http.ts b/packages/ai/src/auth-gateway/http.ts index 21ea80d61..ec905a308 100644 --- a/packages/ai/src/auth-gateway/http.ts +++ b/packages/ai/src/auth-gateway/http.ts @@ -5,19 +5,50 @@ * and peer-resolution logic. */ import { timingSafeEqual as nodeTimingSafeEqual } from "node:crypto"; +import type { Api, AssistantMessage, Model } from "../types"; const JSON_HEADERS = { "Content-Type": "application/json", "X-Content-Type-Options": "nosniff", } as const; -export function json(status: number, body: unknown): Response { +export function json(status: number, body: unknown, headers?: Record): Response { return new Response(JSON.stringify(body) ?? "null", { status, - headers: JSON_HEADERS, + headers: headers ? { ...JSON_HEADERS, ...headers } : JSON_HEADERS, }); } +/** + * Diagnostic response headers for translated inference requests, mirroring the + * names existing gateway-aware clients already parse: `x-request-id` / + * `request-id` (surfaced as `_request_id` by the OpenAI and Anthropic SDKs, + * matches the gateway log line), LiteLLM's model-resolution and cost headers, + * and OpenAI's `openai-processing-ms`. Model/request-id headers are always + * present; `message` — the final assistant message, available only on + * non-streaming responses — adds the computed cost, and `startedAt` the wall + * time. Streaming responses send headers before usage exists, so they carry + * only the identity headers. + */ +export function gatewayResponseHeaders( + model: Model, + info: { requestId: string; message?: AssistantMessage; startedAt?: number }, +): Record { + const headers: Record = { + "x-request-id": info.requestId, + "request-id": info.requestId, + "x-litellm-model-id": model.id, + }; + if (model.baseUrl) headers["x-litellm-model-api-base"] = model.baseUrl; + if (info.message) headers["x-litellm-response-cost"] = info.message.usage.cost.total.toString(); + if (info.startedAt !== undefined) { + const elapsed = (performance.now() - info.startedAt).toFixed(0); + headers["x-litellm-response-duration-ms"] = elapsed; + headers["openai-processing-ms"] = elapsed; + } + return headers; +} + export function resolvePeer(req: Request): string { const fwd = req.headers.get("x-forwarded-for"); if (fwd) return fwd.split(",")[0].trim(); @@ -165,6 +196,8 @@ const CORS_HEADERS: Record = { "Access-Control-Allow-Methods": "GET, POST, OPTIONS", "Access-Control-Allow-Headers": "authorization, content-type, anthropic-version, anthropic-beta, openai-organization, openai-project, x-stainless-*, x-api-key", + "Access-Control-Expose-Headers": + "x-request-id, request-id, x-litellm-model-id, x-litellm-model-api-base, x-litellm-response-cost, x-litellm-response-duration-ms, openai-processing-ms", "Access-Control-Max-Age": "86400", }; diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index ae4d47415..2624dd028 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -32,7 +32,15 @@ import { completeSimple, streamSimple } from "../stream"; import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types"; import { deterministicUuid } from "../utils/deterministic-id"; import { parseBind } from "../utils/parse-bind"; -import { captureRequestHeaders, corsHeaders, isAuthorized, json, resolvePeer, withCors } from "./http"; +import { + captureRequestHeaders, + corsHeaders, + gatewayResponseHeaders, + isAuthorized, + json, + resolvePeer, + withCors, +} from "./http"; import type { AuthGatewayServerHandle, AuthGatewayServerOptions, @@ -334,6 +342,8 @@ async function handleFormatEndpoint( req: Request, peer: string, ): Promise { + const startedAt = performance.now(); + const requestId = crypto.randomUUID(); const controller = mirrorRequestAbort(req); if (controller.signal.aborted) return clientClosedResponse(route); @@ -430,6 +440,7 @@ async function handleFormatEndpoint( ); logger.info("auth-gateway request", { + requestId, format: route.label, model: parsed.modelId, resolvedProvider: model.provider, @@ -458,7 +469,11 @@ async function handleFormatEndpoint( const classified = classifyGatewayError(errorMessage); return route.module.formatError(classified.status, classified.type, errorMessage); } - return json(200, route.module.encodeResponse(message, parsed.modelId)); + return json( + 200, + route.module.encodeResponse(message, parsed.modelId), + gatewayResponseHeaders(model, { requestId, message, startedAt }), + ); } catch (error) { if (controller.signal.aborted) return clientClosedResponse(route); const classified = classifyGatewayError(error); @@ -493,6 +508,7 @@ async function handleFormatEndpoint( return new Response(sseStream, { status: 200, headers: { + ...gatewayResponseHeaders(model, { requestId }), "Content-Type": "text/event-stream; charset=utf-8", "Cache-Control": "no-cache", Connection: "keep-alive", @@ -519,6 +535,8 @@ async function handleFormatEndpoint( * path. */ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise { + const startedAt = performance.now(); + const requestId = crypto.randomUUID(); const controller = mirrorRequestAbort(req); const aborted = (): Response => piNative.formatError(499, "request_aborted", "client closed request"); if (controller.signal.aborted) return aborted(); @@ -605,6 +623,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe streamOpts.sessionId ??= sessionId; logger.info("auth-gateway request", { + requestId, format: "pi-native", model: parsed.modelId, resolvedProvider: model.provider, @@ -633,7 +652,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe const classified = classifyGatewayError(errorMessage); return piNative.formatError(classified.status, classified.type, errorMessage); } - return json(200, { message }); + return json(200, { message }, gatewayResponseHeaders(model, { requestId, message, startedAt })); } catch (error) { if (controller.signal.aborted) return aborted(); const classified = classifyGatewayError(error); @@ -664,6 +683,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe return new Response(sseStream, { status: 200, headers: { + ...gatewayResponseHeaders(model, { requestId }), "Content-Type": "text/event-stream; charset=utf-8", "Cache-Control": "no-cache", Connection: "keep-alive", diff --git a/packages/ai/src/providers/google-auth.ts b/packages/ai/src/providers/google-auth.ts index 7aabaf5cc..740a253cf 100644 --- a/packages/ai/src/providers/google-auth.ts +++ b/packages/ai/src/providers/google-auth.ts @@ -13,7 +13,6 @@ */ import { Buffer } from "node:buffer"; -import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils"; @@ -329,22 +328,3 @@ export function __resetVertexTokenCache(): void { tokenCache.clear(); inflight.clear(); } - -/** - * Sync best-effort probe for a usable Vertex bearer credential source — an explicit access-token - * env var, `GOOGLE_APPLICATION_CREDENTIALS`, a user ADC file, or a GCP runtime whose metadata - * server can mint ADC (GCE/Cloud Run/App Engine/Functions). Lets callers prefer the bearer - * Interactions transport only when ADC is actually reachable, without paying the async - * metadata-probe cost for API-key-only setups. - */ -export function hasVertexBearerCredentialsHint(): boolean { - if (Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN) return true; - if (Bun.env.GOOGLE_APPLICATION_CREDENTIALS) return true; - // GCP-hosted runtimes expose ADC via the metadata server; these env vars mark those runtimes. - if (Bun.env.K_SERVICE || Bun.env.FUNCTION_TARGET || Bun.env.GAE_ENV || Bun.env.GCE_METADATA_HOST) return true; - try { - return fs.existsSync(userAdcPath()); - } catch { - return false; - } -} diff --git a/packages/ai/src/providers/google-interactions.ts b/packages/ai/src/providers/google-interactions.ts deleted file mode 100644 index 3f5612809..000000000 --- a/packages/ai/src/providers/google-interactions.ts +++ /dev/null @@ -1,753 +0,0 @@ -import { parseGeminiModel } from "@oh-my-pi/pi-catalog/identity"; -import { calculateCost } from "@oh-my-pi/pi-catalog/models"; -import { fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils"; -import * as AIError from "../error"; -import type { - AssistantMessage, - Context, - FetchImpl, - ImageContent, - Message, - Model, - ProviderSessionState, - TextContent, - ToolCall, - Usage, -} from "../types"; -import { shouldSendServiceTier } from "../types"; -import { normalizeSystemPrompts } from "../utils"; -import { AssistantMessageEventStream } from "../utils/event-stream"; -import { convertTools, type GoogleSharedStreamOptions, type GoogleThinkingLevel } from "./google-shared"; - -type GoogleInteractionsApi = "google-generative-ai" | "google-vertex"; -type GoogleInteractionsModel = Model; -type GoogleOptions = GoogleSharedStreamOptions; - -const GOOGLE_INTERACTIONS_STATE_KEY = "google-interactions-state"; - -/** Provider session state storing the last Gemini Interactions response id. */ -export interface GoogleInteractionsProviderSessionState extends ProviderSessionState { - lastInteractionId?: string; -} - -/** Conversation anchor for continuing an Interactions turn from a prior assistant response. */ -export interface InteractionAnchor { - id?: string; - messageIndex?: number; -} - -type InteractionContent = { type: "text"; text: string } | { type: "image"; data: string; mime_type: string }; - -interface InteractionUserInputStep { - type: "user_input"; - content: InteractionContent[]; -} - -interface InteractionModelOutputStep { - type: "model_output"; - content: InteractionContent[]; -} - -interface InteractionFunctionCallStep { - type: "function_call"; - id: string; - name: string; - arguments: Record; -} - -interface InteractionFunctionResultStep { - type: "function_result"; - name: string; - call_id: string; - result: InteractionContent[]; - is_error?: boolean; -} - -interface InteractionThoughtStep { - type: "thought"; - summary?: InteractionContent[]; - signature?: string; -} - -interface InteractionThoughtSummaryDelta { - type: "thought_summary"; - content?: InteractionContent; -} - -interface InteractionThoughtSignatureDelta { - type: "thought_signature"; - signature?: string; -} - -interface InteractionArgumentsDelta { - type: "arguments_delta"; - arguments?: string; -} - -interface PendingInteractionToolCall { - id: string; - name: string; - argumentsText: string; - argumentsObject: Record; -} - -type InteractionInputStep = - | InteractionUserInputStep - | InteractionModelOutputStep - | InteractionFunctionCallStep - | InteractionFunctionResultStep; - -type InteractionStep = - | InteractionModelOutputStep - | InteractionFunctionCallStep - | InteractionFunctionResultStep - | InteractionThoughtStep - | InteractionUserInputStep; - -type InteractionThinkingLevel = "minimal" | "low" | "medium" | "high"; - -interface InteractionGenerationConfig { - temperature?: number; - top_p?: number; - top_k?: number; - min_p?: number; - presence_penalty?: number; - frequency_penalty?: number; - repetition_penalty?: number; - max_output_tokens?: number; - thinking_level?: InteractionThinkingLevel; - thinking_budget?: number; -} - -interface GoogleInteractionRequest { - model: string; - input: InteractionInputStep[]; - stream: true; - previous_interaction_id?: string; - system_instruction?: string; - tools?: { functionDeclarations: Record[] }[]; - store?: boolean; - generation_config?: InteractionGenerationConfig; - service_tier?: string; -} - -interface InteractionUsage { - total_input_tokens?: number; - total_cached_tokens?: number; - total_output_tokens?: number; - total_thought_tokens?: number; - total_tokens?: number; -} - -interface InteractionResource { - id?: string; - status?: string; - usage?: InteractionUsage; -} - -interface InteractionStreamMetadata { - total_usage?: InteractionUsage; -} - -interface InteractionSseEvent { - event_type?: string; - index?: number; - step?: InteractionStep; - delta?: - | InteractionContent - | InteractionFunctionCallStep - | InteractionThoughtStep - | InteractionThoughtSummaryDelta - | InteractionThoughtSignatureDelta - | InteractionArgumentsDelta; - interaction?: InteractionResource; - interaction_id?: string; - status?: string; - metadata?: InteractionStreamMetadata; - error?: { message?: string; code?: string | number }; -} - -function emptyUsage(): Usage { - return { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }; -} - -function getGoogleInteractionsState( - providerSessionState: Map | undefined, - create: boolean, -): GoogleInteractionsProviderSessionState | undefined { - if (!providerSessionState) return undefined; - const existing = providerSessionState.get(GOOGLE_INTERACTIONS_STATE_KEY) as - | GoogleInteractionsProviderSessionState - | undefined; - if (existing || !create) return existing; - const state: GoogleInteractionsProviderSessionState = { close: () => {} }; - providerSessionState.set(GOOGLE_INTERACTIONS_STATE_KEY, state); - return state; -} - -function interactionContentFromText(text: string): InteractionContent[] { - return text.length === 0 ? [] : [{ type: "text", text }]; -} - -function interactionContentFromParts(parts: readonly (TextContent | ImageContent)[]): InteractionContent[] { - const content: InteractionContent[] = []; - for (const part of parts) { - if (part.type === "text") { - if (part.text.length > 0) content.push({ type: "text", text: part.text }); - } else { - content.push({ type: "image", data: part.data, mime_type: part.mimeType }); - } - } - return content; -} - -function userInputStepFromMessage(message: Extract): InteractionUserInputStep { - const content = - typeof message.content === "string" - ? interactionContentFromText(message.content) - : interactionContentFromParts(message.content); - return { type: "user_input", content }; -} - -function functionResultStepFromMessage( - message: Extract, -): InteractionFunctionResultStep { - const result = interactionContentFromParts(message.content); - return { - type: "function_result", - name: message.toolName, - call_id: message.toolCallId, - result: result.length > 0 ? result : [{ type: "text", text: "" }], - ...(message.isError ? { is_error: true } : {}), - }; -} - -function appendAssistantInteractionSteps(message: AssistantMessage, steps: InteractionInputStep[]): void { - let modelContent: InteractionContent[] = []; - const flushModelContent = (): void => { - if (modelContent.length === 0) return; - steps.push({ type: "model_output", content: modelContent }); - modelContent = []; - }; - - for (const block of message.content) { - if (block.type === "text") { - if (block.text.length > 0) modelContent.push({ type: "text", text: block.text }); - } else if (block.type === "toolCall") { - flushModelContent(); - steps.push({ type: "function_call", id: block.id, name: block.name, arguments: block.arguments }); - } - } - flushModelContent(); -} - -function interactionMessagesAfterAnchor( - messages: readonly Message[], - anchorIndex: number | undefined, -): readonly Message[] { - return anchorIndex === undefined ? messages : messages.slice(anchorIndex + 1); -} - -function buildInteractionInput(context: Context, anchorIndex: number | undefined): InteractionInputStep[] { - const input: InteractionInputStep[] = []; - for (const message of interactionMessagesAfterAnchor(context.messages, anchorIndex)) { - if (message.role === "user" || message.role === "developer") { - const step = userInputStepFromMessage(message); - if (step.content.length > 0) input.push(step); - } else if (message.role === "toolResult") { - input.push(functionResultStepFromMessage(message)); - } else if (anchorIndex === undefined) { - appendAssistantInteractionSteps(message, input); - } - } - return input.length > 0 ? input : [{ type: "user_input", content: [{ type: "text", text: "" }] }]; -} - -function toInteractionThinkingLevel(level: GoogleThinkingLevel): InteractionThinkingLevel | undefined { - switch (level) { - case "MINIMAL": - return "minimal"; - case "LOW": - return "low"; - case "MEDIUM": - return "medium"; - case "HIGH": - return "high"; - case "THINKING_LEVEL_UNSPECIFIED": - return undefined; - } -} - -function buildInteractionGenerationConfig(options: GoogleOptions | undefined): InteractionGenerationConfig | undefined { - const config: InteractionGenerationConfig = {}; - if (options?.temperature !== undefined) config.temperature = options.temperature; - if (options?.topP !== undefined) config.top_p = options.topP; - if (options?.topK !== undefined) config.top_k = options.topK; - if (options?.minP !== undefined) config.min_p = options.minP; - if (options?.presencePenalty !== undefined) config.presence_penalty = options.presencePenalty; - if (options?.frequencyPenalty !== undefined) config.frequency_penalty = options.frequencyPenalty; - if (options?.repetitionPenalty !== undefined) config.repetition_penalty = options.repetitionPenalty; - if (options?.maxTokens !== undefined) config.max_output_tokens = options.maxTokens; - if (options?.thinking?.level !== undefined) { - const thinkingLevel = toInteractionThinkingLevel(options.thinking.level); - if (thinkingLevel !== undefined) config.thinking_level = thinkingLevel; - } else if (options?.thinking?.budgetTokens !== undefined) { - config.thinking_budget = options.thinking.budgetTokens; - } - return Object.keys(config).length > 0 ? config : undefined; -} - -function buildInteractionRequest( - model: GoogleInteractionsModel, - context: Context, - options: GoogleOptions | undefined, - anchor: InteractionAnchor, -): GoogleInteractionRequest { - const systemInstruction = normalizeSystemPrompts(context.systemPrompt).join("\n\n"); - const generationConfig = buildInteractionGenerationConfig(options); - return { - model: model.id, - input: buildInteractionInput(context, anchor.messageIndex), - stream: true, - ...(anchor.id !== undefined ? { previous_interaction_id: anchor.id } : {}), - ...(systemInstruction.length > 0 ? { system_instruction: systemInstruction } : {}), - ...(context.tools && context.tools.length > 0 ? { tools: convertTools(context.tools, model) } : {}), - ...(options?.storeInteraction !== undefined ? { store: options.storeInteraction } : {}), - ...(generationConfig !== undefined ? { generation_config: generationConfig } : {}), - ...(shouldSendServiceTier(options?.serviceTier, model.provider) ? { service_tier: options?.serviceTier } : {}), - }; -} - -function applyInteractionUsage( - model: GoogleInteractionsModel, - output: AssistantMessage, - usage: InteractionUsage, -): void { - const thinkingTokens = usage.total_thought_tokens ?? 0; - output.usage = { - input: (usage.total_input_tokens ?? 0) - (usage.total_cached_tokens ?? 0), - output: (usage.total_output_tokens ?? 0) + thinkingTokens, - cacheRead: usage.total_cached_tokens ?? 0, - cacheWrite: 0, - totalTokens: usage.total_tokens ?? 0, - ...(thinkingTokens > 0 ? { reasoningTokens: thinkingTokens } : {}), - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }; - calculateCost(model, output.usage); -} - -function parseInteractionFunctionCall(value: unknown): InteractionFunctionCallStep | undefined { - if (!value || typeof value !== "object" || Array.isArray(value)) return undefined; - const record = value as Record; - if (record.type !== "function_call") return undefined; - if (typeof record.id !== "string" || typeof record.name !== "string") return undefined; - const args = record.arguments; - return { - type: "function_call", - id: record.id, - name: record.name, - arguments: args && typeof args === "object" && !Array.isArray(args) ? { ...args } : {}, - }; -} - -function pendingToolCallFromStep(call: InteractionFunctionCallStep): PendingInteractionToolCall { - return { - id: call.id, - name: call.name, - argumentsText: "", - argumentsObject: call.arguments, - }; -} - -function parseInteractionArguments(text: string): Record { - if (text.trim().length === 0) return {}; - try { - const parsed: unknown = JSON.parse(text); - if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return { ...parsed }; - } catch { - return {}; - } - return {}; -} - -/** Provider-specific URL, headers, and fetch implementation for an Interactions request. */ -export interface GoogleInteractionsPlan { - url: string; - headers: Record; - fetch?: FetchImpl; -} - -/** - * Streams Gemini Interactions API model-mode responses for direct Google and Vertex providers. - * - * `fallback`, when supplied, is the legacy `:streamGenerateContent` stream factory. It runs - * transparently — forwarding its events into this stream — when the Interactions attempt fails - * before any content is emitted with a signal that the model/endpoint does not support - * Interactions (HTTP 404/400). Provide it only for auto-selected Interactions requests so an - * explicit `useInteractionsApi: true` still surfaces failures. - */ -export function streamGoogleInteractions(args: { - model: Model; - context: Context; - options: GoogleSharedStreamOptions | undefined; - api: T; - anchor: InteractionAnchor; - state: GoogleInteractionsProviderSessionState | undefined; - prepare: () => GoogleInteractionsPlan | Promise; - fallback?: () => AssistantMessageEventStream; -}): AssistantMessageEventStream { - const { model, context, options, anchor, state } = args; - const stream = new AssistantMessageEventStream(); - const output: AssistantMessage = { - role: "assistant", - content: [], - api: args.api, - provider: model.provider, - model: model.id, - usage: emptyUsage(), - stopReason: "stop", - timestamp: Date.now(), - }; - const storeInteraction = options?.storeInteraction !== false; - - void (async () => { - let started = false; - let sawTerminal = false; - let currentTextBlock: TextContent | undefined; - let currentThinkingBlock: Extract | undefined; - let pendingThinkingSignature: string | undefined; - const stepKinds = new Map(); - const pendingToolCalls = new Map(); - const ensureStarted = (): void => { - if (started) return; - stream.push({ type: "start", partial: output }); - started = true; - }; - const endOpenBlocks = (): void => { - if (currentTextBlock) { - stream.push({ - type: "text_end", - contentIndex: output.content.indexOf(currentTextBlock), - content: currentTextBlock.text, - partial: output, - }); - currentTextBlock = undefined; - } - if (currentThinkingBlock) { - stream.push({ - type: "thinking_end", - contentIndex: output.content.indexOf(currentThinkingBlock), - content: currentThinkingBlock.thinking, - partial: output, - }); - currentThinkingBlock = undefined; - } - }; - const emitText = (text: string): void => { - if (text.length === 0) return; - ensureStarted(); - if (!currentTextBlock) { - if (currentThinkingBlock) endOpenBlocks(); - currentTextBlock = { type: "text", text: "" }; - output.content.push(currentTextBlock); - stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output }); - } - currentTextBlock.text += text; - stream.push({ - type: "text_delta", - contentIndex: output.content.indexOf(currentTextBlock), - delta: text, - partial: output, - }); - }; - const applyThinkingSignature = (signature: string | undefined): void => { - if (!signature) return; - if (currentThinkingBlock) { - currentThinkingBlock.thinkingSignature = signature; - } else { - pendingThinkingSignature = signature; - } - }; - const emitThinking = (text: string): void => { - if (text.length === 0) return; - ensureStarted(); - if (!currentThinkingBlock) { - if (currentTextBlock) endOpenBlocks(); - currentThinkingBlock = { type: "thinking", thinking: "", thinkingSignature: pendingThinkingSignature }; - pendingThinkingSignature = undefined; - output.content.push(currentThinkingBlock); - stream.push({ type: "thinking_start", contentIndex: output.content.length - 1, partial: output }); - } - currentThinkingBlock.thinking += text; - stream.push({ - type: "thinking_delta", - contentIndex: output.content.indexOf(currentThinkingBlock), - delta: text, - partial: output, - }); - }; - const emitToolCall = (call: InteractionFunctionCallStep): void => { - ensureStarted(); - endOpenBlocks(); - const toolCall: ToolCall = { - type: "toolCall", - id: call.id, - name: call.name, - arguments: call.arguments, - }; - output.content.push(toolCall); - const contentIndex = output.content.length - 1; - stream.push({ type: "toolcall_start", contentIndex, partial: output }); - stream.push({ - type: "toolcall_delta", - contentIndex, - delta: JSON.stringify(toolCall.arguments), - partial: output, - }); - stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); - }; - const emitPendingToolCall = (pending: PendingInteractionToolCall): void => { - emitToolCall({ - type: "function_call", - id: pending.id, - name: pending.name, - arguments: - pending.argumentsText.length > 0 - ? parseInteractionArguments(pending.argumentsText) - : pending.argumentsObject, - }); - }; - - let prepared = false; - try { - const plan = await args.prepare(); - prepared = true; - let requestBody: unknown = buildInteractionRequest(model, context, options, anchor); - const replacement = await options?.onPayload?.(requestBody, model); - if (replacement !== undefined) requestBody = replacement; - const response = await fetchWithRetry(() => plan.url, { - method: "POST", - headers: { - ...plan.headers, - "Content-Type": "application/json", - Accept: "text/event-stream", - }, - body: JSON.stringify(requestBody), - signal: options?.signal, - fetch: plan.fetch, - }); - if (!response.ok) { - const errorText = await response.text().catch(() => ""); - throw new AIError.GoogleApiError( - `Google Interactions API error (${response.status}): ${errorText}`, - response.status, - { - headers: response.headers, - }, - ); - } - if (!response.body) { - throw new AIError.ProviderResponseError("Google Interactions API returned an empty response body", { - provider: model.provider, - kind: "empty-body", - }); - } - for await (const event of readSseJson(response.body, options?.signal, sse => - options?.onSseEvent?.({ event: sse.event, data: sse.data, raw: [...sse.raw] }, model), - )) { - if (event.error) { - throw new AIError.ProviderResponseError(event.error.message ?? "Google Interactions API stream error", { - provider: model.provider, - kind: "runtime", - }); - } - if (event.metadata?.total_usage) applyInteractionUsage(model, output, event.metadata.total_usage); - if (event.event_type === "interaction.created") { - if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; - } else if (event.event_type === "step.start" && event.index !== undefined && event.step) { - stepKinds.set(event.index, event.step.type); - const call = parseInteractionFunctionCall(event.step); - if (call) { - pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); - } else if (event.step.type === "thought") { - applyThinkingSignature(event.step.signature); - for (const item of event.step.summary ?? []) { - if (item.type === "text") emitThinking(item.text); - } - } - } else if (event.event_type === "step.delta" && event.index !== undefined && event.delta) { - const call = parseInteractionFunctionCall(event.delta); - if (call) { - pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); - } else if (event.delta.type === "text") { - if (stepKinds.get(event.index) === "thought") emitThinking(event.delta.text); - else emitText(event.delta.text); - } else if (event.delta.type === "thought_summary") { - if (event.delta.content?.type === "text") emitThinking(event.delta.content.text); - } else if (event.delta.type === "thought_signature") { - applyThinkingSignature(event.delta.signature); - } else if (event.delta.type === "arguments_delta") { - const pending = pendingToolCalls.get(event.index); - if (pending && event.delta.arguments) pending.argumentsText += event.delta.arguments; - } - } else if (event.event_type === "step.stop" && event.index !== undefined) { - const stepKind = stepKinds.get(event.index); - const pending = pendingToolCalls.get(event.index); - if (pending) { - emitPendingToolCall(pending); - pendingToolCalls.delete(event.index); - } else { - endOpenBlocks(); - if (stepKind === "thought") pendingThinkingSignature = undefined; - } - stepKinds.delete(event.index); - } else if (event.event_type === "interaction.completed" || event.event_type === "interaction.complete") { - if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; - if (event.interaction?.usage) applyInteractionUsage(model, output, event.interaction.usage); - for (const pending of pendingToolCalls.values()) emitPendingToolCall(pending); - pendingToolCalls.clear(); - endOpenBlocks(); - output.stopReason = - event.interaction?.status === "requires_action" || - output.content.some(block => block.type === "toolCall") - ? "toolUse" - : "stop"; - if (storeInteraction && state) state.lastInteractionId = output.responseId; - sawTerminal = true; - ensureStarted(); - stream.push({ type: "done", reason: output.stopReason, message: output }); - } - } - if (!sawTerminal) { - throw new AIError.ProviderResponseError("Google Interactions API stream ended without a terminal event", { - provider: model.provider, - kind: "incomplete-stream", - }); - } - } catch (error) { - // Auto-selected Interactions degrades to `:streamGenerateContent` when no content has - // streamed yet and the failure means Interactions can't serve this request — the bearer - // credential couldn't be resolved (prepare threw) or the model/endpoint rejected it - // (HTTP 404/400). Provider 401/403/429/5xx still surface. Mirrors the OpenAI Responses - // `previous_response_id` fallback. - const unsupported = error instanceof AIError.GoogleApiError && (error.status === 404 || error.status === 400); - if (!started && args.fallback && !options?.signal?.aborted && (!prepared || unsupported)) { - for await (const event of args.fallback()) stream.push(event); - return; - } - output.stopReason = options?.signal?.aborted ? "aborted" : "error"; - output.errorMessage = error instanceof Error ? error.message : String(error); - stream.push({ type: "error", reason: output.stopReason, error: output }); - } - })(); - - return stream; -} - -function findAssistantInteractionAnchor( - context: Context, - interactionId: string, - provider: string, -): InteractionAnchor | undefined { - for (let index = context.messages.length - 1; index >= 0; index -= 1) { - const message = context.messages[index]; - if (message?.role === "assistant" && message.provider === provider && message.responseId === interactionId) { - return { id: interactionId, messageIndex: index }; - } - } - return undefined; -} - -function latestAssistantInteractionAnchor(context: Context, provider: string): InteractionAnchor | undefined { - for (let index = context.messages.length - 1; index >= 0; index -= 1) { - const message = context.messages[index]; - if (message?.role === "assistant" && message.provider === provider && message.responseId) { - return { id: message.responseId, messageIndex: index }; - } - } - return undefined; -} - -function resolveInteractionAnchor( - context: Context, - explicitPreviousInteractionId: string | undefined, - state: GoogleInteractionsProviderSessionState | undefined, - provider: string, -): InteractionAnchor { - if (explicitPreviousInteractionId !== undefined) { - return ( - findAssistantInteractionAnchor(context, explicitPreviousInteractionId, provider) ?? { - id: explicitPreviousInteractionId, - } - ); - } - const lineageAnchor = latestAssistantInteractionAnchor(context, provider); - if (lineageAnchor) return lineageAnchor; - if (state?.lastInteractionId) - return findAssistantInteractionAnchor(context, state.lastInteractionId, provider) ?? {}; - return {}; -} - -/** - * Whether a model is served by the Gemini Interactions API. Interactions is a Gemini 3-era - * transport, so the catalog subset that supports it is Gemini 3.0+. Older Gemini and non-Gemini - * ids keep `:streamGenerateContent`, which covers the full catalog. - */ -export function modelSupportsInteractions(model: Pick): boolean { - const parsed = parseGeminiModel(model.id); - return parsed !== null && parsed.version.major >= 3; -} - -/** - * Resolves whether a Google provider call should use Interactions and which lineage anchor to send. - * - * Precedence: explicit `useInteractionsApi: false` always wins (force generateContent); otherwise - * Interactions engages when explicitly requested, when continuing a stored interaction - * (`previousInteractionId`/assistant lineage/session state), or when `autoEligible` (the - * zero-config default for the capable model subset on the official endpoint). `auto` flags the - * last case for the caller — it is the only mode that wires up the generateContent fallback. - */ -export function resolveInteractionDispatch(args: { - context: Context; - options: GoogleSharedStreamOptions | undefined; - provider: string; - autoEligible: boolean; -}): { - useInteractions: boolean; - auto: boolean; - anchor: InteractionAnchor; - state: GoogleInteractionsProviderSessionState | undefined; -} { - const explicitPreviousInteractionId = args.options?.previousInteractionId; - if (args.options?.storeInteraction === false && explicitPreviousInteractionId !== undefined) { - throw new AIError.ConfigurationError( - "Google Interactions API cannot combine storeInteraction:false with previousInteractionId.", - ); - } - const explicitOptOut = args.options?.useInteractionsApi === false; - const explicitOptIn = args.options?.useInteractionsApi === true || explicitPreviousInteractionId !== undefined; - const storageEnabled = args.options?.storeInteraction !== false; - const existingState = storageEnabled - ? getGoogleInteractionsState(args.options?.providerSessionState, false) - : undefined; - const anchor = explicitOptOut - ? {} - : resolveInteractionAnchor(args.context, explicitPreviousInteractionId, existingState, args.provider); - const useInteractions = !explicitOptOut && (explicitOptIn || anchor.id !== undefined || args.autoEligible); - const auto = useInteractions && !explicitOptIn; - const interactionState = - useInteractions && storageEnabled - ? getGoogleInteractionsState( - args.options?.providerSessionState, - args.options?.providerSessionState !== undefined, - ) - : undefined; - return { useInteractions, auto, anchor, state: interactionState }; -} diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index 28c805b41..2559af6d4 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -79,19 +79,6 @@ export interface GoogleSharedStreamOptions extends StreamOptions { hideThinkingSummary?: boolean; /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ serviceTier?: ServiceTier; - /** - * Continues a Gemini Interactions API conversation from a stored interaction. - * When set on the direct Google provider, the request uses `/interactions` - * with `previous_interaction_id` instead of the legacy generateContent stream. - */ - previousInteractionId?: string; - /** - * Uses the Gemini Interactions API for direct Google requests, storing the - * returned interaction id on the assistant response for follow-up turns. - */ - useInteractionsApi?: boolean; - /** Overrides Interactions API request storage; default is the API default (`true`). */ - storeInteraction?: boolean; } /** diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 66b1868b9..647c2c8d5 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -2,13 +2,7 @@ import { $env } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { getVertexAccessToken, hasVertexBearerCredentialsHint } from "./google-auth"; -import { - type GoogleInteractionsPlan, - modelSupportsInteractions, - resolveInteractionDispatch, - streamGoogleInteractions, -} from "./google-interactions"; +import { getVertexAccessToken } from "./google-auth"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -22,126 +16,86 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions { } const API_VERSION = "v1"; -const INTERACTIONS_API_VERSION = "v1beta1"; -const INTERACTIONS_API_REVISION = "2026-05-20"; export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( model: Model<"google-vertex">, context: Context, options?: GoogleVertexOptions, ): AssistantMessageEventStream => { - const runGenerateContent = (): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-vertex", - retainTextSignature: true, - prepare: async (): Promise => { - const apiKey = resolveApiKey(options); - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - params.config ||= {}; - if (!params.config.safetySettings) { - params.config.safetySettings = [ - { - category: "HARM_CATEGORY_HATE_SPEECH", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_DANGEROUS_CONTENT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_HARASSMENT", - threshold: "OFF", - }, - ]; - } - const baseHeaders: Record = { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; - // Vertex AI ignores a `serviceTier` request-body field (unlike the direct - // Gemini API); priority must travel as a request header. Only `priority` - // has a documented Vertex request control — `flex` has none, so it's a no-op. - if (options?.serviceTier === "priority") { - baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; - } - - if (apiKey) { - // Explicit `location` is a deliberate residency choice: honor it and let - // a 404 surface. An ambient env-derived region falls back to the global - // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working - // global-only request. - const explicitLocation = options?.location; - const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; - const host = resolveEndpointHost(location); - const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; - const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; - return { - params, - url: `https://${host}/${path}`, - fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, - headers: { - ...baseHeaders, - "x-goog-api-key": apiKey, - }, - fetch: options?.fetch, - }; - } - - const project = resolveProject(options); - const location = resolveLocation(options); - const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); - const host = resolveEndpointHost(location); - const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; - return { - params, - url, - headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, - fetch: options?.fetch, - }; - }, - }); - - // Default Gemini 3+ onto Interactions whenever a bearer credential source exists (ADC file, - // `GOOGLE_APPLICATION_CREDENTIALS`, or an explicit access-token env). Interactions needs bearer - // auth, so express API-key-only setups stay on generateContent — and an express key, when - // present, still serves the generateContent fallback. Interactions always targets the official - // global `aiplatform` host; the fallback also recovers ids the endpoint rejects. - const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ - context, - options, - provider: model.provider, - autoEligible: modelSupportsInteractions(model) && hasVertexBearerCredentialsHint(), - }); - if (!useInteractions) return runGenerateContent(); - - return streamGoogleInteractions({ + return streamGoogleGenAI({ model, - context, options, api: "google-vertex", - anchor, - state, - prepare: async (): Promise => { + retainTextSignature: true, + prepare: async (): Promise => { + const apiKey = resolveApiKey(options); + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + params.config ||= {}; + if (!params.config.safetySettings) { + params.config.safetySettings = [ + { + category: "HARM_CATEGORY_HATE_SPEECH", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_DANGEROUS_CONTENT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_HARASSMENT", + threshold: "OFF", + }, + ]; + } + const baseHeaders: Record = { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + // Vertex AI ignores a `serviceTier` request-body field (unlike the direct + // Gemini API); priority must travel as a request header. Only `priority` + // has a documented Vertex request control — `flex` has none, so it's a no-op. + if (options?.serviceTier === "priority") { + baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; + } + + if (apiKey) { + // Explicit `location` is a deliberate residency choice: honor it and let + // a 404 surface. An ambient env-derived region falls back to the global + // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working + // global-only request. + const explicitLocation = options?.location; + const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + const host = resolveEndpointHost(location); + const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; + return { + params, + url: `https://${host}/${path}`, + fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, + headers: { + ...baseHeaders, + "x-goog-api-key": apiKey, + }, + fetch: options?.fetch, + }; + } + const project = resolveProject(options); + const location = resolveLocation(options); const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); + const host = resolveEndpointHost(location); + const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { - url: `https://aiplatform.googleapis.com/${INTERACTIONS_API_VERSION}/projects/${project}/locations/global/interactions`, - headers: { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - Authorization: `Bearer ${accessToken}`, - "Api-Revision": INTERACTIONS_API_REVISION, - }, + params, + url, + headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, fetch: options?.fetch, }; }, - fallback: auto ? runGenerateContent : undefined, }); }; diff --git a/packages/ai/src/providers/google.ts b/packages/ai/src/providers/google.ts index 3bd8ed7ff..f9d38a1f6 100644 --- a/packages/ai/src/providers/google.ts +++ b/packages/ai/src/providers/google.ts @@ -2,7 +2,6 @@ import * as AIError from "../error"; import { getEnvApiKey } from "../stream"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { modelSupportsInteractions, resolveInteractionDispatch, streamGoogleInteractions } from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -27,61 +26,22 @@ export const streamGoogle: StreamFunction<"google-generative-ai"> = ( ); } - const runGenerateContent = (): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-generative-ai", - prepare: (): GoogleGenAIRequestPlan => { - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - // `model.baseUrl` already includes the API version segment when set (mirrors the - // `apiVersion: ""` reset that the SDK relied on for custom base URLs). - const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; - const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; - const headers: Record = { - "x-goog-api-key": apiKey, - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; - return { params, url, headers, fetch: options?.fetch }; - }, - }); - - // Default Gemini 3+ on the official endpoint onto Interactions (custom proxy base URLs keep - // generateContent, which serves the full catalog). The fallback recovers ids the endpoint rejects. - const trimmedBase = model.baseUrl?.trim(); - let officialEndpoint = !trimmedBase; - if (trimmedBase) { - try { - officialEndpoint = new URL(trimmedBase).hostname === "generativelanguage.googleapis.com"; - } catch { - officialEndpoint = false; - } - } - const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ - context, - options, - provider: model.provider, - autoEligible: officialEndpoint && modelSupportsInteractions(model), - }); - if (!useInteractions) return runGenerateContent(); - - return streamGoogleInteractions({ + return streamGoogleGenAI({ model, - context, options, api: "google-generative-ai", - anchor, - state, - prepare: () => ({ - url: `${trimmedBase || DEFAULT_GENERATIVE_LANGUAGE_BASE}/interactions`, - headers: { + prepare: (): GoogleGenAIRequestPlan => { + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + // `model.baseUrl` already includes the API version segment when set (mirrors the + // `apiVersion: ""` reset that the SDK relied on for custom base URLs). + const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; + const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; + const headers: Record = { "x-goog-api-key": apiKey, ...(model.headers ?? {}), ...(options?.headers ?? {}), - }, - fetch: options?.fetch, - }), - fallback: auto ? runGenerateContent : undefined, + }; + return { params, url, headers, fetch: options?.fetch }; + }, }); }; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 7c66e6c9f..1d1dbd45b 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1432,9 +1432,6 @@ function mapOptionsForApi( streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs, streamIdleTimeoutMs: options?.streamIdleTimeoutMs, providerSessionState: options?.providerSessionState, - useInteractionsApi: options?.useInteractionsApi, - storeInteraction: options?.storeInteraction, - previousInteractionId: options?.previousInteractionId, maxInFlightRequests: options?.maxInFlightRequests, onPayload: options?.onPayload, onResponse: options?.onResponse, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 5b568de29..d5903e5dd 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -424,22 +424,6 @@ export interface StreamOptions { providerSessionState?: Map; /** Canonical Codex compaction classification; ignored by other providers. */ codexCompaction?: CodexCompactionRequestContext; - /** - * Force Gemini model-mode Interactions API transport for providers that support it. - * When unset, those providers may still use Interactions to continue known - * server-side conversation lineage via `previousInteractionId` or stored state. - */ - useInteractionsApi?: boolean; - /** - * Whether supported Interactions transports should store server-side conversation - * state and return response ids for follow-up turns. Defaults to true. - */ - storeInteraction?: boolean; - /** - * Explicit Interactions response id to continue. Mutually exclusive with - * `storeInteraction: false` because the follow-up itself must be storable. - */ - previousInteractionId?: string; /** * Optional per-provider concurrent request cap for LLM stream calls. Keys are * provider ids (`model.provider`); positive numeric values cap in-flight diff --git a/packages/ai/test/auth-gateway-response-headers.test.ts b/packages/ai/test/auth-gateway-response-headers.test.ts new file mode 100644 index 000000000..986d5c084 --- /dev/null +++ b/packages/ai/test/auth-gateway-response-headers.test.ts @@ -0,0 +1,103 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry"; +import { startAuthGateway } from "@oh-my-pi/pi-ai/auth-gateway"; +import { AuthStorage } from "@oh-my-pi/pi-ai/auth-storage"; +import { createMockModel, type MockModel, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock"; + +interface GatewayHarness { + url: string; + mock: MockModel; + close(): Promise; +} + +async function bootGateway(): Promise { + registerMockApi(); + const dir = await fs.mkdtemp(path.join(os.tmpdir(), "gw-response-headers-")); + const storage = await AuthStorage.create(path.join(dir, "auth.db")); + storage.setRuntimeApiKey("openrouter", "test-key"); + const mock = createMockModel({ provider: "openrouter", id: "mock/header-model" }); + const handle = startAuthGateway({ + bind: "127.0.0.1:0", + bearerTokens: ["t"], + storage, + resolveModel: () => mock.model, + version: "test", + }); + return { + url: handle.url, + mock, + close: async () => { + await handle.close(); + storage.close(); + await fs.rm(dir, { recursive: true, force: true }); + }, + }; +} + +afterEach(() => { + clearCustomApis(); +}); + +describe("auth-gateway diagnostic response headers", () => { + it("non-streaming responses carry cost, model id, request id, and duration", async () => { + const gw = await bootGateway(); + try { + gw.mock.push({ + content: ["hello"], + usage: { + input: 10, + output: 5, + totalTokens: 15, + cost: { input: 0.001, output: 0.0002, total: 0.0012 }, + }, + }); + const res = await fetch(`${gw.url}/v1/chat/completions`, { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: "Bearer t" }, + body: JSON.stringify({ + model: "mock/header-model", + messages: [{ role: "user", content: "hi" }], + stream: false, + }), + }); + expect(res.status).toBe(200); + expect(res.headers.get("x-litellm-response-cost")).toBe("0.0012"); + expect(res.headers.get("x-litellm-model-id")).toBe("mock/header-model"); + const duration = res.headers.get("x-litellm-response-duration-ms"); + expect(duration).not.toBeNull(); + expect(Number(duration)).toBeGreaterThanOrEqual(0); + expect(res.headers.get("openai-processing-ms")).toBe(duration); + const requestId = res.headers.get("x-request-id"); + expect(requestId).toMatch(/^[0-9a-f-]{36}$/); + expect(res.headers.get("request-id")).toBe(requestId); + } finally { + await gw.close(); + } + }); + + it("streaming responses carry the model and request ids but no cost (unknown at header time)", async () => { + const gw = await bootGateway(); + try { + gw.mock.push({ content: ["hello"] }); + const res = await fetch(`${gw.url}/v1/chat/completions`, { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: "Bearer t" }, + body: JSON.stringify({ + model: "mock/header-model", + messages: [{ role: "user", content: "hi" }], + stream: true, + }), + }); + expect(res.status).toBe(200); + expect(res.headers.get("x-litellm-model-id")).toBe("mock/header-model"); + expect(res.headers.get("x-request-id")).toMatch(/^[0-9a-f-]{36}$/); + expect(res.headers.get("x-litellm-response-cost")).toBeNull(); + await res.text(); + } finally { + await gw.close(); + } + }); +}); diff --git a/packages/ai/test/google-empty-response-retry.test.ts b/packages/ai/test/google-empty-response-retry.test.ts index f45464685..2e66440c1 100644 --- a/packages/ai/test/google-empty-response-retry.test.ts +++ b/packages/ai/test/google-empty-response-retry.test.ts @@ -89,8 +89,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return calls === 1 ? sse(genaiChunk("")) : sse(genaiChunk("Hello!")); }; - // Pin the generateContent transport: gemini-3 ids now auto-route to Interactions by default. - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); const { events, starts } = await drain(stream); const result = await stream.result(); @@ -108,7 +107,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return sse(genaiChunk("")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); const result = await stream.result(); expect(calls).toBe(3); // MAX_EMPTY_STREAM_RETRIES (2) + 1 initial attempt @@ -138,7 +137,6 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, - useInteractionsApi: false, }); const { events } = await drain(stream); const result = await stream.result(); @@ -196,7 +194,6 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, - useInteractionsApi: false, }); const result = await stream.result(); diff --git a/packages/ai/test/google-interactions.test.ts b/packages/ai/test/google-interactions.test.ts deleted file mode 100644 index 136fe2d95..000000000 --- a/packages/ai/test/google-interactions.test.ts +++ /dev/null @@ -1,511 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; -import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; -import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; -import { streamSimple } from "@oh-my-pi/pi-ai/stream"; -import type { AssistantMessage, Context, FetchImpl, Model, Tool, Usage } from "@oh-my-pi/pi-ai/types"; -import { buildModel } from "@oh-my-pi/pi-catalog/build"; - -function googleModel(baseUrl = "https://generativelanguage.googleapis.com/v1beta"): Model<"google-generative-ai"> { - return buildModel({ - id: "gemini-3.5-flash", - name: "Gemini 3.5 Flash", - api: "google-generative-ai", - provider: "google", - baseUrl, - reasoning: true, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1_000_000, - maxTokens: 8_192, - }); -} - -function vertexModel(id = "gemini-3.5-flash"): Model<"google-vertex"> { - return buildModel({ - id, - name: id, - api: "google-vertex", - provider: "google-vertex", - baseUrl: "", - reasoning: true, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1_000_000, - maxTokens: 8_192, - }); -} - -function sseResponse(events: readonly unknown[]): Response { - const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; - return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } }); -} - -const weatherTool: Tool = { - name: "get_weather", - description: "Get weather", - parameters: { - type: "object", - properties: { city: { type: "string" } }, - required: ["city"], - additionalProperties: false, - }, -}; - -describe("Google Interactions API", () => { - it("chains tool results with previous_interaction_id from the prior assistant response", async () => { - const model = googleModel(); - const requestBodies: unknown[] = []; - let calls = 0; - const fetchMock: FetchImpl = async (_input, init) => { - requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); - calls += 1; - if (calls === 1) { - return sseResponse([ - { - event_type: "interaction.created", - interaction: { id: "int_1", status: "in_progress" }, - }, - { event_type: "step.start", index: 0, step: { type: "thought" } }, - { - event_type: "step.delta", - index: 0, - delta: { type: "thought_signature", signature: "thought_sig_1" }, - }, - { - event_type: "step.delta", - index: 0, - delta: { type: "thought_summary", content: { type: "text", text: "Checking weather.\n" } }, - }, - { event_type: "step.stop", index: 0 }, - { - event_type: "step.start", - index: 1, - step: { - type: "function_call", - id: "call_weather", - name: "get_weather", - arguments: {}, - }, - }, - { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: '{"city":"Bos' } }, - { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: 'ton"}' } }, - { event_type: "step.stop", index: 1 }, - { - event_type: "interaction.completed", - interaction: { - id: "int_1", - status: "requires_action", - usage: { total_input_tokens: 10, total_output_tokens: 2, total_tokens: 12 }, - }, - }, - ]); - } - return sseResponse([ - { event_type: "interaction.created", interaction: { id: "int_2", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text: "Sunny." } }, - { event_type: "step.stop", index: 0 }, - { - event_type: "interaction.completed", - interaction: { - id: "int_2", - status: "completed", - usage: { total_input_tokens: 3, total_output_tokens: 1, total_tokens: 4 }, - }, - }, - ]); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const firstContext: Context = { - systemPrompt: ["Use concise weather reports."], - messages: [{ role: "user", content: "Need weather", timestamp: 1 }], - tools: [weatherTool], - }; - const first = await streamGoogle(model, firstContext, { - apiKey: "test-key", - fetch: fetchMock, - useInteractionsApi: true, - thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, - }).result(); - - expect(first.responseId).toBe("int_1"); - expect(first.stopReason).toBe("toolUse"); - expect(first.content).toEqual([ - { type: "thinking", thinking: "Checking weather.\n", thinkingSignature: "thought_sig_1" }, - { type: "toolCall", id: "call_weather", name: "get_weather", arguments: { city: "Boston" } }, - ]); - expect(requestBodies[0]).toMatchObject({ - model: "gemini-3.5-flash", - stream: true, - input: [{ type: "user_input", content: [{ type: "text", text: "Need weather" }] }], - system_instruction: "Use concise weather reports.", - tools: [{ functionDeclarations: [{ name: "get_weather" }] }], - generation_config: { thinking_level: "high" }, - }); - expect(requestBodies[0]).not.toHaveProperty("previous_interaction_id"); - expect(JSON.stringify(requestBodies[0])).not.toContain("thinking_budget"); - - const secondContext: Context = { - messages: [ - { role: "user", content: "Need weather", timestamp: 1 }, - first, - { - role: "toolResult", - toolCallId: "call_weather", - toolName: "get_weather", - content: [{ type: "text", text: "72F and sunny" }], - isError: false, - timestamp: 2, - }, - ], - tools: [weatherTool], - systemPrompt: ["Use concise weather reports."], - }; - const second = await streamGoogle(model, secondContext, { - apiKey: "test-key", - fetch: fetchMock, - thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, - }).result(); - - expect(second.responseId).toBe("int_2"); - expect(second.content).toEqual([{ type: "text", text: "Sunny." }]); - expect(requestBodies[1]).toMatchObject({ - previous_interaction_id: "int_1", - input: [ - { - type: "function_result", - name: "get_weather", - call_id: "call_weather", - result: [{ type: "text", text: "72F and sunny" }], - }, - ], - tools: [{ functionDeclarations: [{ name: "get_weather" }] }], - system_instruction: "Use concise weather reports.", - generation_config: { thinking_level: "high" }, - }); - expect(JSON.stringify(requestBodies[1])).not.toContain("Need weather"); - expect(JSON.stringify(requestBodies[1])).not.toContain("thinking_budget"); - }); - - it("does not expose or reuse interaction ids when storage is disabled", async () => { - const model = googleModel(); - const requestBodies: unknown[] = []; - const fetchMock: FetchImpl = async (_input, init) => { - requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); - return sseResponse([ - { event_type: "interaction.created", interaction: { id: "unstored_int", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text: "Done." } }, - { event_type: "step.stop", index: 0 }, - { event_type: "interaction.completed", interaction: { id: "unstored_int", status: "completed" } }, - ]); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const result = await streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { - apiKey: "test-key", - fetch: fetchMock, - useInteractionsApi: true, - storeInteraction: false, - }, - ).result(); - - expect(result.responseId).toBeUndefined(); - expect(requestBodies[0]).toMatchObject({ store: false }); - expect(() => - streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { - apiKey: "test-key", - fetch: fetchMock, - storeInteraction: false, - previousInteractionId: "unstored_int", - }, - ), - ).toThrow(/storeInteraction:false/); - }); - - it("reads thought payload from step.start without leaking a prior signature", async () => { - const model = googleModel(); - const fetchMock: FetchImpl = async () => - sseResponse([ - { event_type: "interaction.created", interaction: { id: "int_3", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "thought", signature: "stale_sig" } }, - { event_type: "step.stop", index: 0 }, - { - event_type: "step.start", - index: 1, - step: { type: "thought", summary: [{ type: "text", text: "Fresh plan.\n" }] }, - }, - { event_type: "step.stop", index: 1 }, - { event_type: "step.start", index: 2, step: { type: "model_output" } }, - { event_type: "step.delta", index: 2, delta: { type: "text", text: "Answer." } }, - { event_type: "step.stop", index: 2 }, - { event_type: "interaction.completed", interaction: { id: "int_3", status: "completed" } }, - ]); - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const result = await streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { apiKey: "test-key", fetch: fetchMock, useInteractionsApi: true }, - ).result(); - - expect(result.content).toEqual([ - { type: "thinking", thinking: "Fresh plan.\n" }, - { type: "text", text: "Answer." }, - ]); - }); -}); - -function genaiSse(text: string): Response { - return new Response( - `data: ${JSON.stringify({ - candidates: [{ content: { parts: [{ text }] }, finishReason: "STOP" }], - usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, - })}\n\n`, - { status: 200, headers: { "content-type": "text/event-stream" } }, - ); -} - -function interactionsTextSse( - id: string, - text: string, - terminal: "interaction.completed" | "interaction.complete" = "interaction.completed", -): Response { - return sseResponse([ - { event_type: "interaction.created", interaction: { id, status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text } }, - { event_type: "step.stop", index: 0 }, - { - event_type: terminal, - interaction: { - id, - status: "completed", - usage: { total_input_tokens: 10, total_output_tokens: 5, total_tokens: 15 }, - }, - }, - ]); -} - -interface CapturedCall { - url: string; - method: string; - headers: Headers; - body: unknown; -} - -function captureFetch(handler: (url: string) => Response): { fetch: FetchImpl; calls: CapturedCall[] } { - const calls: CapturedCall[] = []; - const fetchMock: FetchImpl = async (input, init) => { - const url = input instanceof Request ? input.url : String(input); - calls.push({ - url, - method: String(init?.method ?? "GET"), - headers: new Headers(init?.headers), - body: init?.body ? JSON.parse(String(init.body)) : undefined, - }); - return handler(url); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - return { fetch: fetchMock, calls }; -} - -const ZERO_USAGE: Usage = { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, -}; - -function assistantWithResponse( - api: "google-vertex" | "google-generative-ai", - provider: string, - responseId: string, -): AssistantMessage { - return { - role: "assistant", - api, - provider, - model: "gemini-3.5-flash", - content: [{ type: "text", text: "prev" }], - usage: ZERO_USAGE, - stopReason: "stop", - timestamp: 2, - responseId, - }; -} - -const userTurn: Context = { messages: [{ role: "user", content: "Hi there", timestamp: 1 }] }; - -describe("Google Interactions API — zero-config default + fallback", () => { - let savedToken: string | undefined; - beforeEach(() => { - // A bearer source makes the Vertex auto-gate fire deterministically on any machine, and - // `getVertexAccessToken` returns it directly (no OAuth/metadata round-trip in tests). - savedToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; - Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = "test-bearer"; - }); - afterEach(() => { - if (savedToken === undefined) delete Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; - else Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = savedToken; - __resetVertexTokenCache(); - }); - - it("auto-routes a capable Vertex model to Interactions under bearer auth", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("vint_1", "Hi")); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls).toHaveLength(1); - expect(calls[0].url).toBe("https://aiplatform.googleapis.com/v1beta1/projects/p/locations/global/interactions"); - expect(calls[0].method).toBe("POST"); - expect(calls[0].headers.get("Api-Revision")).toBe("2026-05-20"); - expect(calls[0].headers.get("Authorization")).toBe("Bearer test-bearer"); - expect(calls[0].body).toMatchObject({ - model: "gemini-3.5-flash", - stream: true, - input: [{ type: "user_input", content: [{ type: "text", text: "Hi there" }] }], - }); - expect(calls[0].body).not.toHaveProperty("contents"); - expect(calls[0].body).not.toHaveProperty("agent"); - expect(calls[0].body).not.toHaveProperty("environment"); - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "Hi" }]); - expect(result.responseId).toBe("vint_1"); - expect(result.usage.totalTokens).toBe(15); - }); - - it("keeps an older (sub-3) Vertex model on generateContent", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogleVertex(vertexModel("gemini-2.5-flash"), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("honors useInteractionsApi:false on a capable Vertex model", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - useInteractionsApi: false, - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("falls back to generateContent when auto Interactions is unsupported (404)", async () => { - const { fetch, calls } = captureFetch(url => - url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("recovered"), - ); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls[0].url).toContain("/interactions"); - expect(calls[1].url).toContain(":streamGenerateContent"); - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "recovered" }]); - }); - - it("surfaces the error (no fallback) when explicit Interactions is unsupported", async () => { - const { fetch, calls } = captureFetch(url => - url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("unexpected"), - ); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - useInteractionsApi: true, - fetch, - }).result(); - - expect(result.stopReason).toBe("error"); - expect(calls.some(c => c.url.includes(":streamGenerateContent"))).toBe(false); - }); - - it("accepts interaction.complete as a terminal-event alias", async () => { - const { fetch } = captureFetch(() => interactionsTextSse("vint_2", "Done", "interaction.complete")); - const result = await streamGoogleVertex(vertexModel(), userTurn, { project: "p", fetch }).result(); - - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "Done" }]); - }); - - it("sends previous_interaction_id only for same-provider assistant lineage", async () => { - const sameProvider = captureFetch(() => interactionsTextSse("vint_3", "ok")); - await streamGoogleVertex( - vertexModel(), - { - messages: [ - { role: "user", content: "a", timestamp: 1 }, - assistantWithResponse("google-vertex", "google-vertex", "vint_prev"), - { role: "user", content: "b", timestamp: 3 }, - ], - }, - { project: "p", fetch: sameProvider.fetch }, - ).result(); - expect(sameProvider.calls[0].body).toMatchObject({ previous_interaction_id: "vint_prev" }); - - const wrongProvider = captureFetch(() => interactionsTextSse("vint_4", "ok")); - await streamGoogleVertex( - vertexModel(), - { - messages: [ - { role: "user", content: "a", timestamp: 1 }, - assistantWithResponse("google-generative-ai", "google", "gint_prev"), - { role: "user", content: "b", timestamp: 3 }, - ], - }, - { project: "p", fetch: wrongProvider.fetch }, - ).result(); - expect(wrongProvider.calls[0].body).not.toHaveProperty("previous_interaction_id"); - }); - - it("auto-routes a capable direct Google model on the official endpoint to Interactions", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("gint_1", "Hi")); - const result = await streamGoogle(googleModel(), userTurn, { apiKey: "k", fetch }).result(); - - expect(calls[0].url).toBe("https://generativelanguage.googleapis.com/v1beta/interactions"); - expect(calls[0].headers.get("x-goog-api-key")).toBe("k"); - expect(result.responseId).toBe("gint_1"); - }); - - it("keeps a custom-baseUrl direct Google model on generateContent", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogle(googleModel("https://proxy.example.com/v1beta"), userTurn, { - apiKey: "k", - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls[0].url.startsWith("https://proxy.example.com/")).toBe(true); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("threads the auto-default through streamSimple for a capable Google model", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("sint_1", "Hi")); - await streamSimple(googleModel(), userTurn, { apiKey: "k", fetch }).result(); - - expect(calls.some(c => c.url.includes("/interactions"))).toBe(true); - }); -}); diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts index 74066c206..4122dcd92 100644 --- a/packages/ai/test/google-service-tier.test.ts +++ b/packages/ai/test/google-service-tier.test.ts @@ -75,9 +75,7 @@ const vertexModel: Model<"google-vertex"> = buildModel({ describe("Google service tier wire encoding", () => { it("Gemini API sends the tier in the request body, not a header", async () => { const { fetch, captured } = capturingFetch(); - await drain( - streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch, useInteractionsApi: false }), - ); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch })); const { headers, body } = captured(); expect(body.serviceTier).toBe("priority"); expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); @@ -89,7 +87,6 @@ describe("Google service tier wire encoding", () => { streamGoogle(geminiModel, context, { apiKey: "k", fetch, - useInteractionsApi: false, thinking: { enabled: true, level: "HIGH" }, hideThinkingSummary: true, }), @@ -119,7 +116,7 @@ describe("Google service tier wire encoding", () => { it("omits the tier entirely when unset", async () => { const { fetch, captured } = capturingFetch(); - await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch, useInteractionsApi: false })); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch })); expect(captured().body.serviceTier).toBeUndefined(); }); }); diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index f5ef242be..1ca718188 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -26,8 +26,6 @@ async function captureGooglePayload( await streamGoogle(model, context, { apiKey: "test-key", - // Capture the generateContent request shape; gemini-3 ids auto-route to Interactions by default. - useInteractionsApi: false, onPayload: payload => { captured = payload as { config: { systemInstruction?: unknown }; contents: unknown[] }; }, diff --git a/packages/ai/test/issue-1270-repro.test.ts b/packages/ai/test/issue-1270-repro.test.ts index 2154ff79a..5d3582254 100644 --- a/packages/ai/test/issue-1270-repro.test.ts +++ b/packages/ai/test/issue-1270-repro.test.ts @@ -44,8 +44,6 @@ describe("issue #1270: Vertex AI global endpoint", () => { const stream = streamGoogleVertex(model, context, { project: "vertex-project", location: "global", - // This asserts the generateContent URL; gemini-3 ids auto-route to Interactions by default. - useInteractionsApi: false, fetch: async input => { const url = input instanceof Request ? input.url : input.toString(); urls.push(url); diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 98580554d..5e800c617 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.4.8", + "version": "16.5.0", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b5669ce7e..05d111a5e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,50 +2,51 @@ ## [Unreleased] +## [16.5.0] - 2026-07-13 + ### Breaking Changes -- Replaced the `--reasoning-slide-*` flag family (`--reasoning-slide-model`, `--reasoning-slide-turns`, `--reasoning-slide-on-action`, `--reasoning-slide-plan`, `--reasoning-slide-plan-at`, `--reasoning-slide-checklist`) with a single downshift mechanism: `--downshift` switches from the starting model to a fast/cheap target at the first completed turn that starts execution — the todo-list init the plan nudge asks for, or any edit/write tool — always with the hidden plan nudge before the switch and the verify-before-finishing checklist after it (the configuration that won benchmark testing). `--downshift-into ` overrides the default "smol"-role target and implies `--downshift`; `--no-downshift` force-disables. The fixed-turn trigger and per-piece plan/checklist toggles are gone. +- Replaced the `--reasoning-slide-*` flag family with a unified `--prewalk` mechanism (`--prewalk`, `--prewalk-into `, and `--no-prewalk`) to manage model handoffs during execution. ### Added -- Added `--downshift` / `--downshift-into ` / `--no-downshift`: start on a strong model, then hand off to a fast/cheap one (default the `smol` role) at the first edit/write tool call *after* the todo list has been initialized. The starting model handles all planning and todo initialization, and begins implementation, before handing off; the fast model includes a verify checklist before finishing. Enable per-user with the `downshift.enabled` setting; force it mid-session with the new `/downshift` slash command, which arms the switch. -- Added display setting to toggle between collapsing or keeping compacted history inline, now applied to live session displays -- Added a compact session-only model picker (Alt+P) for quick model switching without changing roles -- Added `@` search to the Alt+P / `/switch` picker: it lists configured Ctrl+P quick roles in matching segment colors and applies the selected role's model and thinking for the current session. -- Redesigned Agent Hub entries as two-line cards: identity (status glyph, name, agent type, parent when nested) on the left, active model + reasoning level and age right-aligned, with the task description on its own line; dropped the redundant `sub · of Main` noise -- Added a project-scoped `launch` tool for shared long-running services and debuggers, with readiness probes, bounded logs, PTY input, restart policies, and automatic teardown after the last omp instance exits. Gated behind the `launch.enabled` setting (default on); when disabled the tool is withdrawn and the bash prompt drops its "use launch" guidance. -- Added `detached` `launch` starts for standalone services that survive every omp instance and broker shutdown, then reconnect to the next broker for logs and explicit stop. +- Added a new `--prewalk` execution flow (with `--prewalk-into ` and `--no-prewalk` overrides) that starts tasks on a strong model for planning and todo initialization before handing off to a faster, cheaper model for implementation. + - Added a status line annotation for the active prewalk phase (armed or active). + - Added the `tui.scrollbackRebuild` setting to gate the erase-and-replay native scrollback rebuild mechanism (defaults to off). +- Added a display setting to toggle between collapsing or keeping compacted history inline in live session displays. +- Added a compact session-only model picker (Alt+P) for quick model switching, featuring `@` search to quickly list and apply configured quick roles. +- Redesigned Agent Hub entries into a cleaner two-line card layout showing identity, active model, reasoning level, age, and task description. +- Added a project-scoped `launch` tool (gated by `launch.enabled`) for managing shared long-running services and debuggers, featuring readiness probes, bounded logs, PTY input, restart policies, and automatic teardown. +- Added support for `detached` launches, allowing standalone services to survive broker shutdowns and reconnect to subsequent sessions. ### Changed -- Refined the downshift planning instructions to require a super-detailed todo list with one item per concrete step, each specifying its target and verification method -- Updated tangential agent forks to ignore parent session history and focus exclusively on the new request -- Hardened `/tan` fork isolation: the clone's inherited todo list is cleared at fork (parent todo reminders no longer drag the tan back onto the parent's task), the fork notice warns that the parent is concurrently editing the same working directory, and the notice is re-injected after each compaction so the fork boundary survives summarization -- Added visual markers in the transcript for elided tool calls that have no corresponding result -- Updated status event log to prioritize the most recent entries in the display window -- Updated the snapcompact shape preview transcript to use the compact scope format shown to models during compaction. - -### Removed - -- Removed the `--downshift-boomerang` feature and its associated configuration setting -- Removed the unreliable Bing and Yahoo HTML-scraping web search providers +- Updated JSON logs (`--mode json`) to include provider payloads in auto-compaction events. +- Updated tangential agent forks (`/tan`) to ignore parent session history and focus exclusively on the new request, hardening isolation with cleared todo lists and concurrent editing warnings. +- Added visual markers in the transcript for elided tool calls that have no corresponding result. +- Updated the status event log to prioritize the most recent entries in the display window. +- Upgraded `@agentclientprotocol/sdk` to version 1.2.1. ### Fixed -- Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache -- Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan -- Fixed inconsistent history rendering when toggling the display setting for compacted items -- Fixed configured `retry.fallbackChains` never engaging on non-retryable provider errors (e.g. "Cloud Code Assist API returned an empty response"): a hard error on a model covered by a fallback chain now switches to the next candidate instead of failing the turn, while still never backoff-retrying the failing model itself -- Fixed transcript rebuilds (compaction, `/compact`, and toggling history display) repainting content below stale scrollback when collapsing history; rebuilds now correctly clear the scrollback buffer when history is collapsed -- Improved auto-compaction to automatically drop images and elide content when context is tight, and added persistent warning badges to the compaction divider when manual intervention is required -- Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately -- Fixed the downshift plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls -- Fixed launch tool rendering stacking a stale pending header over a bare `✓ Launch` line and raw text: the tool now uses a merged registry renderer with one per-op status header (op, target, `state · pid · uptime` meta), stripped log cursor suffixes, capped collapsed log/list previews, and a launch tool glyph -- Fixed confusing launch start/wait results when readiness timed out with the log pattern already matched (readiness needs log AND port): the result printed a contradictory `Ready: ` next to `Readiness timed out` without naming the failing condition. Daemon snapshots now carry the unmet conditions (`readyPending`), and start/wait results state exactly what never happened (e.g. `port 3100 on 127.0.0.1 never accepted connections`); the TUI shows a `waiting on port` badge on starting daemons -- Fixed the in-process `stat` builtin mangling BSD-style invocations like `stat -f "%Sm %N" file` (macOS muscle memory): GNU `-f` means `--file-system`, so the format string was treated as a file operand — printing filesystem info for the real operands and erroring with `cannot read file system information for '%Sm %N'`. A `-f` whose format value contains `%` is now detected as BSD syntax and translated to the GNU equivalent (`%Sm`→`%y`, `%N`→`%n`, `%z`→`%s`, epoch/`S`-form times, owner/group/permission and `H`/`L` sub-field directives, `-L`/`-n`/`-q`/`-F` flag clusters, with `%n`/`%t` as literal newline/tab); directives with no GNU counterpart fail with a clear `unsupported BSD format directive` error -- Fixed the remaining GNU-flavored shell builtins that broke under macOS/BSD muscle memory, using the same unambiguous-detection approach as the `stat` fix (only invocations that are invalid or nonsensical under GNU semantics are reinterpreted; unsupported BSD forms fail loudly instead of producing wrong output): `date -r ` formats the epoch when no such file exists (GNU `-r FILE` mtime preserved), signed `date -v±N` adjustments translate to `-d` relative dates and `-j` is accepted (`-j -f` strptime parse mode and field-set `-v` error clearly); `sed -i '' 's/…/…/' file` drops the BSD empty backup-suffix token instead of treating it as the script; `mktemp -t prefix` without X's creates `$TMPDIR/prefix.XXXXXXXXXX` (the GNU `too few X's` error path); `tail -r` reverses input by delegating to `tac` (with `-n`/`-c`/`-f` combinations erroring clearly); `find -E` maps to `-regextype posix-extended` ahead of the expression; `base64 -D` decodes as an alias of `-d`; and `ln -sfh` works via a `-h` alias of `--no-dereference` (clap's `-h` help short is dropped to match real GNU/BSD ln; `--help` unchanged) +- Fixed terminal scrollback duplication issues with expanded streaming edit previews (Ctrl+O) by using a viewport-sized tail window. +- Fixed custom model role resolution and alias parsing, ensuring canonical role selectors (`@role`) and thinking suffixes resolve correctly across all configuration surfaces. +- Fixed quadratic growth in JSON logs by eliding redundant message snapshots and payloads. +- Fixed prompt cache misses and incorrect cache key pinning for `/tan` and `/fork` clones. +- Fixed inconsistent history rendering and scrollback repainting when toggling the display setting for compacted items. +- Fixed `retry.fallbackChains` failing to engage on non-retryable provider errors, ensuring the agent correctly falls back to the next candidate model. +- Improved auto-compaction to automatically drop images and elide content when context is tight, and added persistent warning badges when manual intervention is required. +- Fixed backgrounded Bash blocks continuing to repaint with live output; they now freeze with a compact job notice while completion is delivered separately. +- Fixed rendering, status display, and PTY control sequence formatting issues in the `launch` tool. +- Fixed in-process shell builtins (including `stat`, `date`, `sed`, `mktemp`, `tail`, `find`, `base64`, and `ln`) to correctly detect and translate macOS/BSD-style arguments and flags, preventing failures caused by GNU-only assumptions. + +### Removed + +- Removed the `--prewalk-boomerang` feature and its associated configuration setting. +- Removed the unreliable Bing and Yahoo HTML-scraping web search providers. ## [16.4.8] - 2026-07-12 + ### Added - Added a predicate form to the browser run's `wait()` helper: `wait(fn, { timeout?, interval? })` polls the function (sync or async) until truthy and resolves with that value, failing with a named timeout error (deadline clamped under the cell budget so it always beats the opaque whole-cell timeout) instead of Bun's `sleep expects a number` or a whole-cell stall from in-page polling Promises; both `wait` forms now register in the stall diagnosis of cell timeouts diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 57a83ecfa..ccf3dbfb2 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.4.8", + "version": "16.5.0", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index d60d3a9d3..f0a1c0195 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -27,9 +27,9 @@ export interface Args { smol?: string; slow?: string; plan?: string; - downshift?: boolean; - noDownshift?: boolean; - downshiftInto?: string; + prewalk?: boolean; + noPrewalk?: boolean; + prewalkInto?: string; planYolo?: boolean; planYoloInto?: string; maxTime?: number; @@ -236,10 +236,10 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map = { "--plan": (result, value) => { result.plan = value; }, - "--downshift-into": (result, value) => { - result.downshiftInto = value; + "--prewalk-into": (result, value) => { + result.prewalkInto = value; }, "--plan-yolo-into": (result, value) => { result.planYoloInto = value; @@ -281,8 +281,8 @@ export const VALUELESS_FLAGS: ReadonlySet = new Set([ "--no-pty", "--hide-thinking", "--advisor", - "--downshift", - "--no-downshift", + "--prewalk", + "--no-prewalk", "--plan-yolo", "--print", "--print-thoughts", diff --git a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts index d734a7da3..1261b672c 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts @@ -102,24 +102,33 @@ export const shellFixtures: Record = { launch_logs: { label: "Launch", renderer: "launch", - args: { op: "logs", name: "web", lines: 100, follow: true, cursor: 1842, timeout: 30 }, + args: { op: "logs", name: "comp-debug", lines: 100, follow: true, cursor: 233_512, timeout: 30 }, result: { content: [ { type: "text", text: [ - "$ bun run dev", - " VITE v6.0.3 ready in 312 ms", - "", - " ➜ Local: http://localhost:5173/", - " ➜ Network: use --host to expose", - "12:04:11 [vite] hmr update /src/App.tsx", - "12:04:15 [vite] hmr update /src/components/Chart.tsx", - "[web: running; cursor=2210]", + "Breakpoint 1: 3 locations.", + "(lldb) run", + "Process 726 launched: '/tmp/compiler'", + "frame #0: 0x0000000100012f80 compiler`parse_expression", + "[comp-debug: ready; cursor=233797]", ].join("\n"), }, ], - details: { op: "logs", cursor: 2210, timedOut: false, state: "running" }, + details: { + op: "logs", + cursor: 233_797, + timedOut: false, + state: "ready", + terminalRows: [ + "\x1b[0mBreakpoint 1: 3 locations.", + "\x1b[0m(lldb) run", + "\x1b[0mProcess 726 launched: '/tmp/compiler'", + "\x1b[0mframe #0: 0x0000000100012f80 compiler`parse_expression", + "\x1b[0m\x1b[1;38;5;2m(lldb)\x1b[0m ", + ], + }, }, errorResult: { content: [{ type: "text", text: "No daemon named web" }], diff --git a/packages/coding-agent/src/commands/launch.ts b/packages/coding-agent/src/commands/launch.ts index 915d56971..49f4df8ee 100644 --- a/packages/coding-agent/src/commands/launch.ts +++ b/packages/coding-agent/src/commands/launch.ts @@ -34,15 +34,15 @@ export default class Index extends Command { plan: Flags.string({ description: "Plan model for architectural planning (or PI_PLAN_MODEL env)", }), - downshift: Flags.boolean({ + prewalk: Flags.boolean({ description: - "Switch from the active model to a fast/cheap model at the first edit/write after the plan's todo list exists (default off; see downshift.enabled)", + "Switch from the active model to a fast/cheap model at the first edit/write after the plan's todo list exists (default off; see prewalk.enabled)", }), - "no-downshift": Flags.boolean({ - description: "Disable downshift even if downshift.enabled is set", + "no-prewalk": Flags.boolean({ + description: "Disable prewalk even if prewalk.enabled is set", }), - "downshift-into": Flags.string({ - description: 'Target model for downshift (default the "smol" role)', + "prewalk-into": Flags.string({ + description: 'Target model for prewalk (default the "smol" role)', }), "plan-yolo": Flags.boolean({ description: diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 32ed6116f..08fdb5e18 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -37,7 +37,14 @@ import { resolveThinkingLevelForModel, } from "../thinking"; import { isAuthenticated, kNoAuth, type ModelRegistry } from "./model-registry"; -import { MODEL_ROLE_IDS, type ModelRole } from "./model-roles"; +import { + DEFAULT_MODEL_ROLE_ALIAS, + formatModelRoleAlias, + LEGACY_MODEL_ROLE_ALIAS_PREFIX, + MODEL_ROLE_ALIAS_PREFIX, + MODEL_ROLE_IDS, + type ModelRole, +} from "./model-roles"; import type { Settings } from "./settings"; function isKnownProvider(provider: string): provider is KnownProvider { @@ -109,8 +116,7 @@ function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): Co * level / `:auto` sentinel); `base` then has the suffix stripped. Otherwise * `base` is the input. * `minColonIndex` requires the colon to appear strictly after that index — - * role-alias callers pass `PREFIX_MODEL_ROLE.length` so the base is at least - * as long as the `pi/` prefix. + * role-alias callers pass the matched alias prefix length. */ function splitThinkingSuffix( pattern: string, @@ -850,17 +856,32 @@ export function parseModelPattern( ); } -const PREFIX_MODEL_ROLE = "pi/"; const DEFAULT_MODEL_ROLE = "default"; +const MODEL_ROLE_ALIAS_PREFIXES = [MODEL_ROLE_ALIAS_PREFIX, LEGACY_MODEL_ROLE_ALIAS_PREFIX]; -function getModelRoleAlias(value: string): ModelRole | undefined { +function isModelRole(role: string): role is ModelRole { + return (MODEL_ROLE_IDS as string[]).includes(role); +} + +/** + * Minimum colon index for splitting a `:` suffix off a role alias, or + * `undefined` when `value` is not role-alias shaped. Doubles as the slice + * offset of the role name for prefixed aliases (`@role`, `pi/role`); the bare + * `*` default alias returns 0 because its colon sits immediately after the + * one-character token (`*:xhigh`). + */ +function modelRoleAliasPrefixLength(value: string): number | undefined { + if (value === DEFAULT_MODEL_ROLE_ALIAS || value.startsWith(`${DEFAULT_MODEL_ROLE_ALIAS}:`)) return 0; + return MODEL_ROLE_ALIAS_PREFIXES.find(prefix => value.startsWith(prefix))?.length; +} + +function getModelRoleAlias(value: string, settings?: Settings): string | undefined { const normalized = value.trim(); - if (!normalized.startsWith(PREFIX_MODEL_ROLE)) return undefined; + const prefixLength = modelRoleAliasPrefixLength(normalized); + if (prefixLength === undefined) return undefined; - const candidate = normalized.slice(PREFIX_MODEL_ROLE.length); - for (const role of MODEL_ROLE_IDS) { - if (candidate === role) return role; - } + const candidate = normalized === DEFAULT_MODEL_ROLE_ALIAS ? DEFAULT_MODEL_ROLE : normalized.slice(prefixLength); + if (isModelRole(candidate) || settings?.getModelRole(candidate) !== undefined) return candidate; return undefined; } @@ -871,7 +892,14 @@ function normalizeModelPatternList(value: string | string[] | undefined): string } function isSessionInheritedAgentPattern(value: string): boolean { - return value === DEFAULT_MODEL_ROLE || value === `${PREFIX_MODEL_ROLE}${DEFAULT_MODEL_ROLE}` || value === "pi/task"; + return ( + value === DEFAULT_MODEL_ROLE || + value === formatModelRoleAlias(DEFAULT_MODEL_ROLE) || + value === DEFAULT_MODEL_ROLE_ALIAS || + value === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}${DEFAULT_MODEL_ROLE}` || + value === formatModelRoleAlias("task") || + value === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}task` + ); } function shouldInheritDefaultBeforePriority(role: ModelRole): boolean { @@ -904,7 +932,7 @@ function resolveDefaultInheritedPatterns( configuredDefault: string | undefined, roleDefaults: string[], settings: Settings | undefined, - visited: Set, + visited: Set, ): string[] { if (!shouldInheritDefaultBeforePriority(role) || !configuredDefault) return []; @@ -912,13 +940,12 @@ function resolveDefaultInheritedPatterns( for (const pattern of normalizeModelPatternList(configuredDefault)) { const { base: aliasCandidate, level: thinkingLevel } = splitThinkingSuffix( pattern, - PREFIX_MODEL_ROLE.length, + modelRoleAliasPrefixLength(pattern) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length, MAX_THINKING_SUFFIX_OPTIONS, ); - const aliasRole = getModelRoleAlias(aliasCandidate); + const aliasRole = getModelRoleAlias(aliasCandidate, settings); if (aliasRole === role) { - // Self-alias (e.g. modelRoles.default = "pi/smol") would loop back to the - // same unset role; collapse straight to the built-in priority chain. + // Self-alias (e.g. modelRoles.default = "@smol") would loop back to the resolved.push( ...(thinkingLevel ? roleDefaults.map(defaultPattern => `${defaultPattern}:${thinkingLevel}`) @@ -927,8 +954,7 @@ function resolveDefaultInheritedPatterns( continue; } if (aliasRole && !visited.has(aliasRole)) { - // Cross-role alias (e.g. modelRoles.default = "pi/slow"): resolve the - // target role's patterns now so downstream one-layer expanders see + // Cross-role alias (e.g. modelRoles.default = "@slow"): resolve the // concrete model patterns instead of another role alias. const recursed = resolveConfiguredRolePattern(pattern, settings, new Set(visited)); if (recursed && recursed.length > 0) { @@ -944,27 +970,29 @@ function resolveDefaultInheritedPatterns( function resolveConfiguredRolePattern( value: string, settings?: Settings, - visited: Set = new Set(), + visited: Set = new Set(), ): string[] | undefined { const normalized = value.trim(); if (!normalized) return undefined; const { base: aliasCandidate, level: thinkingLevel } = splitThinkingSuffix( normalized, - PREFIX_MODEL_ROLE.length, + modelRoleAliasPrefixLength(normalized) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length, MAX_THINKING_SUFFIX_OPTIONS, ); - const role = getModelRoleAlias(aliasCandidate); + const role = getModelRoleAlias(aliasCandidate, settings); if (!role) return [normalized]; if (visited.has(role)) return undefined; visited.add(role); const configured = settings?.getModelRole(role)?.trim(); const configuredDefault = settings?.getModelRole(DEFAULT_MODEL_ROLE)?.trim(); - const roleDefaults = rolePriorityDefaults(role); + const roleDefaults = isModelRole(role) ? rolePriorityDefaults(role) : []; const resolved = configured ? normalizeModelPatternList(configured) - : resolveDefaultInheritedPatterns(role, configuredDefault, roleDefaults, settings, visited); + : isModelRole(role) + ? resolveDefaultInheritedPatterns(role, configuredDefault, roleDefaults, settings, visited) + : roleDefaults; if (resolved.length === 0) { resolved.push(...roleDefaults); } @@ -976,7 +1004,7 @@ function resolveConfiguredRolePattern( } /** - * Expand a role alias like "pi/smol" to the configured model string. + * Expand a role alias like "@smol" to the configured model string. */ export function expandRoleAlias(value: string, settings?: Settings): string { const normalized = value.trim(); @@ -1014,8 +1042,13 @@ export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOp const singleAgentPattern = normalizedAgentPatterns.length === 1 ? normalizedAgentPatterns[0] : undefined; const agentInheritsSessionModel = singleAgentPattern ? isSessionInheritedAgentPattern(singleAgentPattern) : false; if (configuredAgentPatterns.length > 0) { + if ( + singleAgentPattern === formatModelRoleAlias("task") || + singleAgentPattern === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}task` + ) { + return configuredAgentPatterns; + } if (!agentInheritsSessionModel) return configuredAgentPatterns; - if (singleAgentPattern === "pi/task") return configuredAgentPatterns; } const fallback = @@ -1102,12 +1135,16 @@ export function extractExplicitThinkingSelector( let current = normalized; while (!visited.has(current)) { visited.add(current); - const strictSelector = splitThinkingSuffix(current, PREFIX_MODEL_ROLE.length).level; + const rolePrefixLength = modelRoleAliasPrefixLength(current) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length; + const strictSelector = splitThinkingSuffix(current, rolePrefixLength).level; if (strictSelector) { return strictSelector; } - const maxSelector = splitThinkingSuffix(current, PREFIX_MODEL_ROLE.length, MAX_THINKING_SUFFIX_OPTIONS).level; - if (maxSelector && (current.startsWith(PREFIX_MODEL_ROLE) || !isLiteralModelSelector(current, options))) { + const maxSelector = splitThinkingSuffix(current, rolePrefixLength, MAX_THINKING_SUFFIX_OPTIONS).level; + if ( + maxSelector && + (modelRoleAliasPrefixLength(current) !== undefined || !isLiteralModelSelector(current, options)) + ) { return maxSelector; } const expanded = expandRoleAlias(current, settings).trim(); @@ -1278,7 +1315,7 @@ export function resolveAdvisorRoleSelection( settings: Settings, availableModels: Model[], ): { model: Model; thinkingLevel?: ConfiguredThinkingLevel } | undefined { - const resolved = resolveModelRoleValue(`${PREFIX_MODEL_ROLE}advisor`, availableModels, { + const resolved = resolveModelRoleValue(formatModelRoleAlias("advisor"), availableModels, { settings, matchPreferences: getModelMatchPreferences(settings), }); @@ -1300,6 +1337,7 @@ export async function resolveModelScope( patterns: string[], modelRegistry: Pick, preferences?: ModelMatchPreferences, + settings?: Settings, ): Promise { const availableModels = modelRegistry.getAvailable(); const context = buildPreferenceContext(availableModels, preferences); @@ -1337,6 +1375,25 @@ export async function resolveModelScope( continue; } + // Role aliases (`@smol`, `pi/slow`) resolve to the role's single concrete + // model — not its whole fallback chain — so a role contributes one scope + // entry exactly like `--model` would pick. (Bare `*` stays a match-all + // glob above; scope semantics, not the default-role alias.) + if (settings && modelRoleAliasPrefixLength(pattern) !== undefined) { + const resolved = resolveModelRoleValue(pattern, availableModels, { settings, matchPreferences: preferences }); + if (resolved.warning) logger.warn(resolved.warning); + if (!resolved.model) { + logger.warn(`No models match pattern "${pattern}"`); + continue; + } + if (resolved.thinkingLevel === AUTO_THINKING) { + addScopedModel(resolved.model, undefined, false); + } else { + addScopedModel(resolved.model, resolved.thinkingLevel, resolved.explicitThinkingLevel); + } + continue; + } + const { model, thinkingLevel, warning, explicitThinkingLevel } = parseModelPatternWithContext( pattern, availableModels, @@ -1386,7 +1443,7 @@ export async function resolveAllowedModels( if (!patterns || patterns.length === 0) { return available; } - const scoped = await resolveModelScope(patterns, modelRegistry, preferences); + const scoped = await resolveModelScope(patterns, modelRegistry, preferences, settings); if (scoped.length === 0) { return []; } @@ -1415,6 +1472,7 @@ export async function resolveAllowedModels( export function filterAvailableModelsByEnabledPatterns( available: Model[], patterns: readonly string[], + settings?: Settings, ): Model[] { if (patterns.length === 0) return available; @@ -1432,6 +1490,13 @@ export function filterAvailableModelsByEnabledPatterns( continue; } + // Mirror resolveModelScope: role aliases resolve to the role's model. + if (settings && modelRoleAliasPrefixLength(pattern) !== undefined) { + const { model } = resolveModelRoleValue(pattern, available, { settings }); + if (model) addAllowed(model); + continue; + } + const { model } = parseModelPatternWithContext(pattern, available, context); if (model) { addAllowed(model); @@ -1456,9 +1521,10 @@ export function resolveCliModel(options: { cliProvider?: string; cliModel?: string; modelRegistry: CliModelRegistry; + settings?: Settings; preferences?: ModelMatchPreferences; }): ResolveCliModelResult { - const { cliProvider, cliModel, modelRegistry, preferences } = options; + const { cliProvider, cliModel, modelRegistry, settings, preferences } = options; if (!cliModel) { return { model: undefined, selector: undefined, warning: undefined, error: undefined }; @@ -1474,6 +1540,19 @@ export function resolveCliModel(options: { }; } + if (!cliProvider && modelRoleAliasPrefixLength(cliModel) !== undefined) { + const resolved = resolveModelRoleValue(cliModel, availableModels, { settings, matchPreferences: preferences }); + if (resolved.model) { + return { + model: resolved.model, + selector: formatModelString(resolved.model), + thinkingLevel: resolved.thinkingLevel, + warning: resolved.warning, + error: undefined, + }; + } + } + const providerMap = new Map(); for (const model of availableModels) { providerMap.set(model.provider.toLowerCase(), model.provider); diff --git a/packages/coding-agent/src/config/model-roles.ts b/packages/coding-agent/src/config/model-roles.ts index 3ab5e9489..6d97d94b4 100644 --- a/packages/coding-agent/src/config/model-roles.ts +++ b/packages/coding-agent/src/config/model-roles.ts @@ -5,6 +5,20 @@ import { isValidThemeColor, type ThemeColor } from "../modes/theme/theme"; import type { Settings } from "./settings"; +/** Canonical prefix for a configured model role selector. */ +export const MODEL_ROLE_ALIAS_PREFIX = "@"; + +/** Legacy prefix accepted for backwards-compatible role selectors. */ +export const LEGACY_MODEL_ROLE_ALIAS_PREFIX = "pi/"; + +/** Shorthand selector for the default model role. */ +export const DEFAULT_MODEL_ROLE_ALIAS = "*"; + +/** Format a model role as its canonical selector. */ +export function formatModelRoleAlias(role: string): string { + return `${MODEL_ROLE_ALIAS_PREFIX}${role}`; +} + export type ModelRole = | "default" | "smol" diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 5888b7ce0..b5be79ab8 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -117,7 +117,7 @@ export const TAB_METADATA: Record = { appearance: ["Theme", "Status Line", "Display", "Images"], - model: ["Thinking", "Sampling", "Prompt", "Retry & Fallback", "Advisor", "Downshift", "Vision"], + model: ["Thinking", "Sampling", "Prompt", "Retry & Fallback", "Advisor", "Prewalk", "Vision"], interaction: [ "Input", "Approvals", @@ -422,15 +422,15 @@ export const SETTINGS_SCHEMA = { "Pair a second model (assigned to the 'advisor' role) that passively reviews each turn and injects notes.", }, }, - "downshift.enabled": { + "prewalk.enabled": { type: "boolean", default: false, ui: { tab: "model", - group: "Downshift", - label: "Enable Downshift", + group: "Prewalk", + label: "Enable Prewalk", description: - "Start on the active model, then switch to a fast/cheap model (default the 'smol' role) at the first edit/write after the plan nudge's todo list exists — the strong model plans, commits the todos, and starts the implementation before handing off. Overridable per session with --downshift / --no-downshift.", + "Start on the active model, then switch to a fast/cheap model (default the 'smol' role) at the first edit/write after the plan nudge's todo list exists — the strong model plans, commits the todos, and starts the implementation before handing off. Overridable per session with --prewalk / --no-prewalk.", }, }, "advisor.subagents": { @@ -897,6 +897,17 @@ export const SETTINGS_SCHEMA = { description: "Remove the 1-character horizontal padding from the left and right of the terminal output", }, }, + "tui.scrollbackRebuild": { + type: "boolean", + default: false, + ui: { + tab: "appearance", + group: "Display", + label: "Rewrite Scrollback", + description: + "Erase and replay terminal scrollback when a block's final form replaces its live preview. When off (default), stale preview copies remain in history and the final content is appended below.", + }, + }, "display.shimmer": { type: "enum", @@ -2639,14 +2650,14 @@ export const SETTINGS_SCHEMA = { group: "Mnemopi", label: "Mnemopi LLM Mode", description: - "Use no LLM, the online tiny model (the TINY role from /models, else pi/smol), or a remote OpenAI-compatible endpoint", + "Use no LLM, the online tiny model (the TINY role from /models, else @smol), or a remote OpenAI-compatible endpoint", condition: "mnemopiActive", options: [ { value: "none", label: "None", description: "Disable Mnemopi LLM-backed extraction" }, { value: "smol", label: "Online (tiny)", - description: "Use the online tiny model (the TINY role from /models, else pi/smol)", + description: "Use the online tiny model (the TINY role from /models, else @smol)", }, { value: "remote", label: "Remote", description: "Use the Mnemopi remote LLM settings below" }, ], @@ -4600,7 +4611,7 @@ export const SETTINGS_SCHEMA = { group: "Tiny Model", label: "Tiny Model", description: - "Session-title model: online (the TINY role from /models, else pi/smol) by default, or a local on-device model", + "Session-title model: online (the TINY role from /models, else @smol) by default, or a local on-device model", options: TINY_TITLE_MODEL_OPTIONS, }, }, diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 14682a59e..cd11ab3c5 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -423,22 +423,25 @@ function formatStreamingDiff( cache?: RenderedStringCache, ): string { if (!diff) return ""; - // Clamp the collapsed tail to the viewport so a tall or fast-growing diff - // cannot outgrow the live window. Otherwise its mutating tail scrolls above - // the native-scrollback commit boundary and the engine re-commits a fresh - // snapshot every streamed frame, stacking duplicate "… more lines above" - // previews in history. The budget is VISUAL rows (a long wrapped line counts + // Clamp the tail to the viewport so a tall or fast-growing diff cannot + // outgrow the live window. Otherwise its mutating rows scroll above the + // native-scrollback commit boundary mid-stream and freeze into immutable + // history as a stale preview snapshot; the finalize repair then recommits + // the final render below it — a duplicated block on the tape. Collapsed + // gets a short fixed tail; expanded widens it to the viewport-sized window, + // never unbounded. The budget is VISUAL rows (a long wrapped line counts // for more than one) at the framed block's inner width (border only — - // contentPaddingLeft is 0); only the visible suffix is syntax-colored, so the - // cheap raw-line wrap walk keeps the per-chunk cost bounded. innerWidth/budget - // are in the cache salt so a resize re-slices. + // contentPaddingLeft is 0); only the visible suffix is syntax-colored, so + // the cheap raw-line wrap walk keeps the per-chunk cost bounded. + // innerWidth/budget are in the cache salt so a resize re-slices. const innerWidth = Math.max(1, width - 2); - const budget = expanded ? Number.POSITIVE_INFINITY : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); + const budget = expanded ? previewWindowRows() : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${innerWidth}:${budget}`, diff, () => { // "Cursor" tail window: pin the last rows to the bottom so freshly streamed // changes stay on screen. The whole-file diff is recomputed every chunk and // its Myers alignment is not monotonic in payload length, so a hunk-aware - // window stutters as rows move between hunks. Expanded lifts the cap. + // window stutters as rows move between hunks. Expanded widens the window + // to the viewport; the full diff appears once the result finalizes. const allLines = diff.replace(/\n+$/u, "").split("\n"); let visualUsed = 0; let cut = allLines.length; diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 320c84fb2..bff2e44d1 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -28,7 +28,7 @@ const taskAgent = { systemPrompt: "Run the task.", source: "bundled", spawns: "*", - model: ["pi/task"], + model: ["@task"], } satisfies AgentDefinition; const reviewerAgent = { @@ -36,7 +36,7 @@ const reviewerAgent = { description: "Reviewer agent", systemPrompt: "Review the task.", source: "bundled", - model: ["pi/smol"], + model: ["@smol"], } satisfies AgentDefinition; interface SessionOptions { diff --git a/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts index 0b4422021..59ad6c844 100644 --- a/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts @@ -193,7 +193,7 @@ describe("runEvalCompletion", () => { expect(resolved).toEqual(["p/smol", "p/default", "p/slow"]); }); - it("prefers the session active model for the default tier, falling back to pi/default", async () => { + it("prefers the session active model for the default tier, falling back to @default", async () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); const session = makeSession({ available: [SMOL, DEFAULT, SLOW], activeModel: "p/slow" }); diff --git a/packages/coding-agent/src/eval/completion-bridge.ts b/packages/coding-agent/src/eval/completion-bridge.ts index 52f04f254..03bf6d778 100644 --- a/packages/coding-agent/src/eval/completion-bridge.ts +++ b/packages/coding-agent/src/eval/completion-bridge.ts @@ -37,9 +37,9 @@ const STRUCTURED_TOOL_NAME = "respond"; type CompletionTier = "smol" | "default" | "slow"; const TIER_TO_PATTERN: Record = { - smol: "pi/smol", - default: "pi/default", - slow: "pi/slow", + smol: "@smol", + default: "@default", + slow: "@slow", }; const completionArgsSchema = type({ @@ -62,7 +62,7 @@ export interface EvalCompletionResult { /** * Resolve a tier to a concrete {@link Model}. `default` prefers the session's - * active model and falls back to the `pi/default` role; `smol`/`slow` resolve + * active model and falls back to the `@default` role; `smol`/`slow` resolve * their respective role patterns. Returns `undefined` when nothing matches. */ function resolveTierModel(tier: CompletionTier, session: ToolSession): Model | undefined { diff --git a/packages/coding-agent/src/extensibility/extensions/model-api.ts b/packages/coding-agent/src/extensibility/extensions/model-api.ts index 3d9560c7e..6a6aeb075 100644 --- a/packages/coding-agent/src/extensibility/extensions/model-api.ts +++ b/packages/coding-agent/src/extensibility/extensions/model-api.ts @@ -25,7 +25,7 @@ export function createExtensionModelQuery( return { list: () => modelRegistry.getAvailable(), current: () => getModel(), - // resolveModelRoleValue expands a role alias (`pi/slow`) to its full configured + // resolveModelRoleValue expands a role alias (`@slow`) to its full configured // priority list and tries each pattern — the same path core selection uses — so a // fallback model lower in the list still resolves. Plain model strings pass through // as a single pattern. diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index ee079659b..97a1fccbf 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -391,7 +391,7 @@ export interface ExtensionModelQuery { /** The current session model, if one is set. */ current(): Model | undefined; /** - * Resolve a model string (`provider/id`, bare id) or role alias (`pi/slow`, a + * Resolve a model string (`provider/id`, bare id) or role alias (`@slow`, a * configured role) to a Model, using the same settings-backed aliases and match * preferences as core selection. Thinking/routing suffixes are accepted and resolved * to the base model (pass effort separately). Returns undefined when nothing matches. diff --git a/packages/coding-agent/src/launch/broker.ts b/packages/coding-agent/src/launch/broker.ts index c2bc6ef5c..9511ae9a9 100644 --- a/packages/coding-agent/src/launch/broker.ts +++ b/packages/coding-agent/src/launch/broker.ts @@ -4,13 +4,15 @@ import * as os from "node:os"; import * as path from "node:path"; import { Process, type PtyRunResult, PtySession } from "@oh-my-pi/pi-natives"; import { isEexist, isEnoent, logger, postmortem, sanitizeText } from "@oh-my-pi/pi-utils"; -import { truncateHead, truncateTail } from "../session/streaming-output"; +import { truncateHead, truncateHeadBytes, truncateTail, truncateTailBytes } from "../session/streaming-output"; import { workerEnvFromParent } from "../subprocess/worker-client"; import { daemonBrokerEndpoint } from "./paths"; import { hasLiveDaemonProjectPresence } from "./presence"; import { DAEMON_IDLE_GRACE_ENV, DAEMON_PROJECT_DIR_ENV, + DAEMON_PTY_COLUMNS, + DAEMON_PTY_ROWS, DAEMON_RUNTIME_DIR_ENV, type DaemonOperation, type DaemonReadySpec, @@ -74,6 +76,11 @@ interface BrokerLease { instanceId: string; } +interface DaemonLogRead { + text: string; + terminalText: string; +} + function quoteShellArg(value: string): string { return `'${value.replaceAll("'", `'\\''`)}'`; } @@ -141,8 +148,7 @@ class DaemonLog { return new DaemonLog(logPath, previousPath, file, file.writer()); } - append(raw: string): string { - const text = sanitizeText(raw); + append(text: string): string { if (text.length === 0 || this.#closed) return text; const bytes = Buffer.byteLength(text, "utf8"); this.#queue = this.#queue.then(async () => { @@ -154,7 +160,7 @@ class DaemonLog { return text; } - async read(head: boolean, lines: number, grep?: string): Promise { + async read(head: boolean, lines: number, grep?: string): Promise { await this.#queue; await this.#writer.flush(); return DaemonLog.readFiles(this.#path, this.#previousPath, head, lines, grep); @@ -167,19 +173,19 @@ class DaemonLog { await this.#writer.end(); } - static async readDir(dir: string, head: boolean, lines: number, grep?: string): Promise { - return DaemonLog.readFiles(path.join(dir, LOG_FILE), path.join(dir, PREVIOUS_LOG_FILE), head, lines, grep); - } - static async readFiles( logPath: string, previousPath: string, head: boolean, lines: number, grep?: string, - ): Promise { + ): Promise { const [previous, current] = await Promise.all([fileTextSlice(previousPath, head), fileTextSlice(logPath, head)]); - let text = sanitizeText(`${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`); + const combined = `${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`; + const terminalText = head + ? truncateHeadBytes(combined, LOG_READ_BYTES).text + : truncateTailBytes(combined, LOG_READ_BYTES).text; + let text = sanitizeText(terminalText); if (grep) { let pattern: RegExp; try { @@ -193,7 +199,10 @@ class DaemonLog { .join("\n"); } const options = { maxLines: lines, maxBytes: 256 * 1024 }; - return head ? truncateHead(text, options).content : truncateTail(text, options).content; + return { + text: head ? truncateHead(text, options).content : truncateTail(text, options).content, + terminalText, + }; } async #rotate(): Promise { @@ -527,8 +536,8 @@ class DaemonBroker { command, cwd: record.spec.cwd, env: workerEnvFromParent({ TERM: "xterm-256color", ...record.spec.env }), - cols: 120, - rows: 40, + cols: DAEMON_PTY_COLUMNS, + rows: DAEMON_PTY_ROWS, shell, }, (error, chunk) => { @@ -626,9 +635,10 @@ class DaemonBroker { #onOutput(record: ManagedDaemon, generation: number, raw: string): void { if (generation !== record.generation) return; - const text = record.log?.append(raw) ?? sanitizeText(raw); + const output = raw.toWellFormed(); + const text = record.log?.append(output) ?? output; record.snapshot.outputBytes += Buffer.byteLength(text, "utf8"); - this.#trackOutput(record, generation, text); + this.#trackOutput(record, generation, sanitizeText(text)); } async #readDetachedOutput(record: ManagedDaemon, generation: number): Promise { @@ -753,13 +763,20 @@ class DaemonBroker { timedOut = !changed; } const lines = Math.max(1, Math.min(1_000, Math.floor(operation.lines))); - const text = record.log + const output = record.log ? await record.log.read(operation.head, lines, operation.grep) - : await DaemonLog.readDir(record.dir, operation.head, lines, operation.grep); + : await DaemonLog.readFiles( + path.join(record.dir, LOG_FILE), + path.join(record.dir, PREVIOUS_LOG_FILE), + operation.head, + lines, + operation.grep, + ); return { op: "logs", name: record.snapshot.name, - text, + text: output.text, + terminalText: record.spec.pty && operation.grep === undefined ? output.terminalText : undefined, cursor: record.snapshot.outputBytes, timedOut, state: record.snapshot.state, diff --git a/packages/coding-agent/src/launch/protocol.ts b/packages/coding-agent/src/launch/protocol.ts index 97bf0631c..5ea71b2a6 100644 --- a/packages/coding-agent/src/launch/protocol.ts +++ b/packages/coding-agent/src/launch/protocol.ts @@ -4,6 +4,10 @@ /** Hidden CLI selector used to re-enter the daemon broker worker. */ export const DAEMON_BROKER_WORKER_ARG = "__omp_worker_daemon_broker"; +/** Fixed dimensions negotiated with every supervised PTY. */ +export const DAEMON_PTY_COLUMNS = 120; +export const DAEMON_PTY_ROWS = 40; + /** Environment key carrying the broker's canonical project directory. */ export const DAEMON_PROJECT_DIR_ENV = "OMP_DAEMON_PROJECT_DIR"; @@ -97,6 +101,8 @@ export type DaemonRpcResult = op: "logs"; name: string; text: string; + /** Raw PTY byte stream used only to reconstruct the terminal screen. */ + terminalText?: string; cursor: number; timedOut: boolean; state: DaemonState; @@ -349,6 +355,8 @@ export function parseDaemonRpcResult(operation: DaemonOperation, value: unknown) op: "logs", name: stringValue(source.name, "result.name"), text: typeof source.text === "string" ? source.text : "", + terminalText: + source.terminalText === undefined ? undefined : rawString(source.terminalText, "result.terminalText"), cursor: numberValue(source.cursor, "result.cursor"), timedOut: booleanValue(source.timedOut, "result.timedOut"), state: daemonState(source.state), diff --git a/packages/coding-agent/src/launch/terminal-output.ts b/packages/coding-agent/src/launch/terminal-output.ts new file mode 100644 index 000000000..da75c5c6c --- /dev/null +++ b/packages/coding-agent/src/launch/terminal-output.ts @@ -0,0 +1,46 @@ +import { logger } from "@oh-my-pi/pi-utils"; +import xterm, { type Terminal as XtermTerminal } from "@xterm/headless"; +import { readTerminalRows } from "../tools/terminal-output"; +import { DAEMON_PTY_COLUMNS, DAEMON_PTY_ROWS } from "./protocol"; + +const VIRTUAL_SCROLLBACK_ROWS = 4_096; + +/** Controls which virtual terminal rows a launch log exposes. */ +export interface TerminalOutputOptions { + head: boolean; + maxRows: number; +} + +function writeTerminal(terminal: XtermTerminal, output: string): Promise { + const { promise, resolve } = Promise.withResolvers(); + terminal.write(output, resolve); + return promise; +} + +/** Replays daemon bytes with the same xterm screen renderer used by PTY mode. */ +export async function renderTerminalOutput( + output: string, + options: TerminalOutputOptions, +): Promise { + if (!output) return []; + const maxRows = Math.max(1, Math.floor(options.maxRows)); + const terminal = new xterm.Terminal({ + cols: DAEMON_PTY_COLUMNS, + rows: DAEMON_PTY_ROWS, + scrollback: Math.max(VIRTUAL_SCROLLBACK_ROWS, maxRows), + allowProposedApi: true, + }); + try { + await writeTerminal(terminal, output); + const rows = readTerminalRows(terminal, 0, terminal.buffer.active.length); + while (rows.at(-1) === "") rows.pop(); + return options.head ? rows.slice(0, maxRows) : rows.slice(-maxRows); + } catch (error) { + logger.debug("Failed to render launch terminal output", { + error: error instanceof Error ? error.message : String(error), + }); + return undefined; + } finally { + terminal.dispose(); + } +} diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index f91f1e31c..9fa982b20 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -852,6 +852,7 @@ export async function buildSessionOptions( cliProvider: parsed.provider, cliModel: parsed.model, modelRegistry, + settings: activeSettings, preferences: modelMatchPreferences, }); if (resolved.warning) { @@ -905,40 +906,40 @@ export async function buildSessionOptions( if (!options.model) options.model = scopedModels[0].model; } - if (parsed.noDownshift && (parsed.downshift || parsed.downshiftInto !== undefined)) { - throw new Error("--no-downshift cannot be combined with --downshift or --downshift-into"); + if (parsed.noPrewalk && (parsed.prewalk || parsed.prewalkInto !== undefined)) { + throw new Error("--no-prewalk cannot be combined with --prewalk or --prewalk-into"); } - const downshiftEnabled = parsed.noDownshift + const prewalkEnabled = parsed.noPrewalk ? false - : parsed.downshift === true || parsed.downshiftInto !== undefined + : parsed.prewalk === true || parsed.prewalkInto !== undefined ? true - : activeSettings.get("downshift.enabled"); - if (downshiftEnabled) { - const rolePattern = expandRoleAlias(parsed.downshiftInto ?? "pi/smol", activeSettings); + : activeSettings.get("prewalk.enabled"); + if (prewalkEnabled) { + const rolePattern = expandRoleAlias(parsed.prewalkInto ?? "@smol", activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.downshiftInto ?? "pi/smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? "@smol"}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); } - options.downshift = { target: resolved.model, thinkingLevel: resolved.thinkingLevel }; + options.prewalk = { target: resolved.model, thinkingLevel: resolved.thinkingLevel }; } if (parsed.planYoloInto !== undefined && !parsed.planYolo) { throw new Error("--plan-yolo-into requires --plan-yolo"); } if (parsed.planYolo) { - const rolePattern = expandRoleAlias(parsed.planYoloInto ?? "pi/smol", activeSettings); + const rolePattern = expandRoleAlias(parsed.planYoloInto ?? "@smol", activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.planYoloInto ?? "pi/smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.planYoloInto ?? "@smol"}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); @@ -1183,6 +1184,7 @@ export async function runRootCommand( modelPatterns, modelRegistry, modelMatchPreferences, + settingsInstance, ); } diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 5b31ec935..3bca86ddc 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -351,12 +351,19 @@ async function elicitFromAcpClient( finish(undefined); }); const response = await promise; - if (response?.action !== "accept" || !response.content) { + if (!isAcceptedElicitation(response) || !response.content) { return undefined; } return response.content.value; } +/** Narrows a `CreateElicitationResponse` to the accepted-with-content branch; the SDK's `action: string` catch-all arm otherwise defeats literal narrowing on `action !== "accept"`. */ +function isAcceptedElicitation( + response: CreateElicitationResponse | undefined, +): response is Extract { + return response?.action === "accept"; +} + /** * Build an {@link ExtensionUIContext} that translates skill/extension UI * requests into ACP elicitations against `connection` for the session diff --git a/packages/coding-agent/src/modes/components/status-line/component.test.ts b/packages/coding-agent/src/modes/components/status-line/component.test.ts index 559a11870..0f5efdeb9 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.test.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.test.ts @@ -4,15 +4,43 @@ import type { AgentSession } from "../../../session/agent-session"; import { getThemeByName, setThemeInstance } from "../../theme/theme"; import { StatusLineComponent } from "./component"; -function makeSessionWithLastMessage(lastMessage: unknown) { +function makeSessionWithLastMessage(lastMessage: unknown, prewalkArmed: boolean = false) { return { - messages: [lastMessage], + messages: lastMessage ? [lastMessage] : [], model: { contextWindow: 128000 }, contextUsageRevision: 0, systemPrompt: [], agent: { state: { tools: [] } }, skills: [], getContextUsage: () => ({ tokens: 42, contextWindow: 128000 }), + state: { + messages: lastMessage ? [lastMessage] : [], + model: { contextWindow: 128000 }, + }, + sessionManager: { + getUsageStatistics: () => ({ + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + orchestrationInput: 0, + orchestrationOutput: 0, + orchestrationCacheRead: 0, + premiumRequests: 0, + cost: 0, + tokensPerSecond: null, + }), + getSessionName: () => "test-session", + }, + getPrewalkState: () => (prewalkArmed ? { target: { id: "cheap-model", provider: "openai" } } : undefined), + getAsyncJobSnapshot: () => undefined, + isAdvisorActive: () => false, + isFastModeActive: () => false, + configuredThinkingLevel: () => undefined, + modelRegistry: { + isUsingOAuth: () => false, + }, }; } @@ -41,4 +69,15 @@ describe("StatusLineComponent", () => { expect(statusLine.getCachedContextBreakdown()).toEqual({ usedTokens: 42, contextWindow: 128000 }); }); + + it("renders Prewalk annotation when prewalk is armed", () => { + const statusLine = new StatusLineComponent(makeSessionWithLastMessage(null, true) as unknown as AgentSession); + + // By default preset, 'mode' segment is included in left/right segments. + // Let's get the border and see if Prewalk is rendered. + const border = statusLine.getTopBorder(100); + // SGR codes might be included, so we check if the stripped content contains "Prewalk" + const stripped = border.content.replace(/\x1b\[[0-9;]*m/g, ""); + expect(stripped).toContain("Prewalk"); + }); }); diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index e709f9d78..f07f30e86 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -1051,6 +1051,10 @@ export class StatusLineComponent implements Component { compactThinkingLevel: this.#resolveSettings().compactThinkingLevel ?? false, planMode: this.#planModeStatus, loopMode: this.#loopModeStatus, + prewalk: + typeof this.session.getPrewalkState === "function" && this.session.getPrewalkState() + ? { enabled: true } + : null, goalMode: this.#goalModeStatus, vibeMode: this.#vibeModeStatus, collab: this.#collabStatus, diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts index 0aad8698e..3ba55ead3 100644 --- a/packages/coding-agent/src/modes/components/status-line/segments.ts +++ b/packages/coding-agent/src/modes/components/status-line/segments.ts @@ -230,6 +230,12 @@ const modeSegment: StatusLineSegment = { return { content: theme.fg(color, content), visible: true }; } + const prewalk = ctx.prewalk; + if (prewalk?.enabled) { + const content = withIcon(theme.icon.prewalk, "Prewalk"); + return { content: theme.fg("accent", content), visible: true }; + } + const goal = ctx.goalMode; if (goal && (goal.enabled || goal.paused)) { return renderGoalMode(ctx, goal); diff --git a/packages/coding-agent/src/modes/components/status-line/types.ts b/packages/coding-agent/src/modes/components/status-line/types.ts index 8fe16380d..06719799b 100644 --- a/packages/coding-agent/src/modes/components/status-line/types.ts +++ b/packages/coding-agent/src/modes/components/status-line/types.ts @@ -60,6 +60,9 @@ export interface SegmentContext { enabled: boolean; paused: boolean; } | null; + prewalk: { + enabled: boolean; + } | null; loopMode: { enabled: boolean; } | null; diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index f28d95ad8..47a277f99 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -473,6 +473,10 @@ export class SelectorController { this.ctx.ui.requestRender(); break; + case "tui.scrollbackRebuild": + this.ctx.ui.setScrollbackRebuild(value as boolean); + break; + case "tui.renderMermaid": setMarkdownMermaidRendering(value as boolean); this.ctx.session.refreshBaseSystemPrompt().catch(err => { diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8311acbc2..50cba2279 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -665,6 +665,7 @@ export class InteractiveMode implements InteractiveModeContext { setMarkdownMermaidRendering(settings.get("tui.renderMermaid")); this.ui = new TUI(new ProcessTerminal(), settings.get("showHardwareCursor")); this.ui.setMaxInlineImages(settings.get("tui.maxInlineImages")); + this.ui.setScrollbackRebuild(settings.get("tui.scrollbackRebuild")); // OSC 66 text-sizing is Kitty-only; resolve the setting against the terminal's // capability (`TERMINAL.textSizing` defaults on for Kitty) so it stays off // unless the user opts in, and never emits raw escapes on other terminals. diff --git a/packages/coding-agent/src/modes/print-mode.test.ts b/packages/coding-agent/src/modes/print-mode.test.ts new file mode 100644 index 000000000..0a5242f22 --- /dev/null +++ b/packages/coding-agent/src/modes/print-mode.test.ts @@ -0,0 +1,71 @@ +/** + * Contract: `--mode json` output stays linear in conversation size. Streaming + * `message_update` events must not re-serialize the in-progress message, and + * printed messages must not carry provider-opaque replay payloads (encrypted + * reasoning history), which previously produced multi-GB transcripts. + */ +import { describe, expect, it } from "bun:test"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import type { AgentSessionEvent } from "../session/agent-session"; +import { printableEvent } from "./print-mode"; + +const assistant: AssistantMessage = { + role: "assistant", + content: [{ type: "text", text: "hello" }], + api: "openai-responses", + provider: "openai", + model: "gpt-test", + usage: {} as AssistantMessage["usage"], + stopReason: "stop", + timestamp: 1, + providerPayload: { type: "openaiResponsesHistory", provider: "openai", dt: true, items: [{ big: "blob" }] }, +}; + +describe("printableEvent", () => { + it("emits only the incremental delta for message_update", () => { + const event: AgentSessionEvent = { + type: "message_update", + message: assistant, + assistantMessageEvent: { type: "text_delta", contentIndex: 0, delta: "hel", partial: assistant }, + }; + const printed = JSON.parse(JSON.stringify(printableEvent(event))); + expect(printed).toEqual({ + type: "message_update", + assistantMessageEvent: { type: "text_delta", contentIndex: 0, delta: "hel" }, + }); + }); + + it("drops the done-variant message snapshot from message_update", () => { + const event: AgentSessionEvent = { + type: "message_update", + message: assistant, + assistantMessageEvent: { type: "done", reason: "stop", message: assistant }, + }; + const printed = JSON.parse(JSON.stringify(printableEvent(event))); + expect(printed).toEqual({ type: "message_update", assistantMessageEvent: { type: "done", reason: "stop" } }); + }); + + it("strips providerPayload from message_end but keeps the message content", () => { + const event: AgentSessionEvent = { type: "message_end", message: assistant }; + const printed = JSON.parse(JSON.stringify(printableEvent(event))) as { + message: Record; + }; + expect(printed.message.providerPayload).toBeUndefined(); + expect(printed.message.content).toEqual([{ type: "text", text: "hello" }]); + expect(printed.message.model).toBe("gpt-test"); + }); + + it("strips providerPayload from every message in agent_end", () => { + const event: AgentSessionEvent = { type: "agent_end", messages: [assistant, assistant] }; + const printed = JSON.parse(JSON.stringify(printableEvent(event))) as { + messages: Array>; + }; + expect(printed.messages).toHaveLength(2); + for (const message of printed.messages) expect(message.providerPayload).toBeUndefined(); + }); + + it("passes unrelated events through untouched", () => { + const event: AgentSessionEvent = { type: "notice", level: "info", message: "hi" }; + expect(printableEvent(event)).toBe(event); + }); +}); diff --git a/packages/coding-agent/src/modes/print-mode.ts b/packages/coding-agent/src/modes/print-mode.ts index dec9d4050..737ba7439 100644 --- a/packages/coding-agent/src/modes/print-mode.ts +++ b/packages/coding-agent/src/modes/print-mode.ts @@ -5,9 +5,10 @@ * - `omp -p "prompt"` - text output * - `omp --mode json "prompt"` - JSON event stream */ +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, ImageContent } from "@oh-my-pi/pi-ai"; import { logger, sanitizeText } from "@oh-my-pi/pi-utils"; -import type { AgentSession } from "../session/agent-session"; +import type { AgentSession, AgentSessionEvent } from "../session/agent-session"; import { isSilentAbort } from "../session/messages"; import { flushTelemetryExport } from "../telemetry-export"; import { initializeExtensions } from "./runtime-init"; @@ -28,6 +29,54 @@ export interface PrintModeOptions { printThoughts?: boolean; } +/** Drop the provider-opaque replay payload (e.g. encrypted reasoning items) before printing. */ +function stripProviderPayload(message: T): T { + if (!("providerPayload" in message) || message.providerPayload === undefined) return message; + const { providerPayload: _providerPayload, ...rest } = message; + return rest as T; +} + +/** + * Shape an event for `--mode json` output. + * + * Removes two classes of bloat so transcripts grow linearly with conversation + * size instead of quadratically (a single long turn used to re-serialize its + * whole in-progress message on every streamed delta, producing multi-GB logs): + * - `message_update` snapshots (`message`, `assistantMessageEvent.partial`, + * and the `done`/`error` payloads) are dropped; only the incremental delta + * is printed. The authoritative message follows in `message_end`. + * - `providerPayload` is transport-native replay state, opaque and useless + * outside this process. + */ +export function printableEvent(event: AgentSessionEvent): unknown { + switch (event.type) { + case "message_update": { + const streamEvent = event.assistantMessageEvent; + if (streamEvent.type === "done" || streamEvent.type === "error") { + return { + type: "message_update", + assistantMessageEvent: { type: streamEvent.type, reason: streamEvent.reason }, + }; + } + const { partial: _partial, ...rest } = streamEvent; + return { type: "message_update", assistantMessageEvent: rest }; + } + case "message_start": + case "message_end": + return { ...event, message: stripProviderPayload(event.message) }; + case "turn_end": + return { + ...event, + message: stripProviderPayload(event.message), + toolResults: event.toolResults.map(stripProviderPayload), + }; + case "agent_end": + return { ...event, messages: event.messages.map(stripProviderPayload) }; + default: + return event; + } +} + /** * Run in print (single-shot) mode. * Sends prompts to the agent and outputs the result. @@ -58,7 +107,7 @@ export async function runPrintMode(session: AgentSession, options: PrintModeOpti session.subscribe(event => { // In JSON mode, output all events if (mode === "json") { - process.stdout.write(`${JSON.stringify(event)}\n`); + process.stdout.write(`${JSON.stringify(printableEvent(event))}\n`); } }); diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index e6ae57aa4..3c64156fc 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -92,6 +92,7 @@ export type SymbolKey = // Icons | "icon.model" | "icon.plan" + | "icon.prewalk" | "icon.goal" | "icon.pause" | "icon.loop" @@ -301,6 +302,7 @@ const UNICODE_SYMBOLS: SymbolMap = { // Icons "icon.model": "⬢", "icon.plan": "🗺", + "icon.prewalk": "🏃", "icon.goal": "🎯", "icon.pause": "⏸", "icon.loop": "↻", @@ -562,6 +564,7 @@ const NERD_SYMBOLS: SymbolMap = { "icon.model": "\uec19", // pick:  | alt:   "icon.plan": "\uf2d2", + "icon.prewalk": "\uf29d", // pick: (nf-fa-bullseye) | alt: (nf-md-target) ◎ ⌖ "icon.goal": "\uf140", // pick: (nf-fa-pause) | alt: ⏸ || @@ -819,6 +822,7 @@ const ASCII_SYMBOLS: SymbolMap = { // Icons "icon.model": "[M]", "icon.plan": "plan", + "icon.prewalk": "prewalk", "icon.goal": "goal", "icon.pause": "||", "icon.loop": "loop", @@ -1819,6 +1823,7 @@ export class Theme { return { model: this.#symbols["icon.model"], plan: this.#symbols["icon.plan"], + prewalk: this.#symbols["icon.prewalk"], goal: this.#symbols["icon.goal"], pause: this.#symbols["icon.pause"], loop: this.#symbols["icon.loop"], diff --git a/packages/coding-agent/src/prompts/agents/designer.md b/packages/coding-agent/src/prompts/agents/designer.md index 72091b563..f45910b69 100644 --- a/packages/coding-agent/src/prompts/agents/designer.md +++ b/packages/coding-agent/src/prompts/agents/designer.md @@ -1,7 +1,7 @@ --- name: designer description: UI/UX specialist for design implementation, review, visual refinement -model: pi/designer +model: "@designer" --- Implement and review UI designs. Edit files, create components, run commands when needed. diff --git a/packages/coding-agent/src/prompts/agents/librarian.md b/packages/coding-agent/src/prompts/agents/librarian.md index d3bb1c40a..dc7764013 100644 --- a/packages/coding-agent/src/prompts/agents/librarian.md +++ b/packages/coding-agent/src/prompts/agents/librarian.md @@ -2,7 +2,7 @@ name: librarian description: Researches external libraries and APIs by reading source code. Returns definitive, source-verified answers. tools: read, grep, glob, bash, lsp, web_search, ast_grep -model: pi/smol +model: "@smol" thinking-level: minimal read-summarize: false output: diff --git a/packages/coding-agent/src/prompts/agents/reviewer.md b/packages/coding-agent/src/prompts/agents/reviewer.md index edb4e51a2..11a468ddf 100644 --- a/packages/coding-agent/src/prompts/agents/reviewer.md +++ b/packages/coding-agent/src/prompts/agents/reviewer.md @@ -3,7 +3,7 @@ name: reviewer description: "Code review specialist for quality/security analysis" tools: read, grep, glob, bash, lsp, web_search, ast_grep spawns: scout -model: pi/slow +model: "@slow" output: properties: overall_correctness: diff --git a/packages/coding-agent/src/prompts/agents/scout.md b/packages/coding-agent/src/prompts/agents/scout.md index 277539b44..97af9b018 100644 --- a/packages/coding-agent/src/prompts/agents/scout.md +++ b/packages/coding-agent/src/prompts/agents/scout.md @@ -2,7 +2,7 @@ name: scout description: MUST be used for exploratory codebase research, rapid code analysis, and broad pattern searches. Fast read-only scout returning compressed context for handoff. tools: read, grep, glob, web_search -model: pi/smol +model: "@smol" thinking-level: medium read-summarize: false output: diff --git a/packages/coding-agent/src/prompts/system/downshift-checklist.md b/packages/coding-agent/src/prompts/system/prewalk-checklist.md similarity index 100% rename from packages/coding-agent/src/prompts/system/downshift-checklist.md rename to packages/coding-agent/src/prompts/system/prewalk-checklist.md diff --git a/packages/coding-agent/src/prompts/system/downshift-continue.md b/packages/coding-agent/src/prompts/system/prewalk-continue.md similarity index 100% rename from packages/coding-agent/src/prompts/system/downshift-continue.md rename to packages/coding-agent/src/prompts/system/prewalk-continue.md diff --git a/packages/coding-agent/src/prompts/system/downshift-plan.md b/packages/coding-agent/src/prompts/system/prewalk-plan.md similarity index 65% rename from packages/coding-agent/src/prompts/system/downshift-plan.md rename to packages/coding-agent/src/prompts/system/prewalk-plan.md index 1a2d4b74c..3f4f63e9e 100644 --- a/packages/coding-agent/src/prompts/system/downshift-plan.md +++ b/packages/coding-agent/src/prompts/system/prewalk-plan.md @@ -8,6 +8,6 @@ First, state the plan itself, explicitly and comprehensively: Be thorough and concrete — this plan is the reference for the remainder of the run. You may verify details with tools after the plan is written, never before. -Then, only once the plan above is complete and detailed, in the SAME reply, capture it as a SUPER-DETAILED todo list (the todo tool): one item per concrete step from the plan — each naming its exact file/symbol/command target and its verification — not a handful of vague phase headings. The todo list must be precise enough that every item can be checked off against an observable result. +Then, only once the plan above is complete, in the SAME reply, capture it as a todo list (the todo tool): 5-9 items, one per MEANINGFUL step, each naming its concrete target and its verification. Only steps that change or verify code belong on the list — no reporting, bookkeeping, cleanup-ceremony, or release-note items. The todo list serves the task, never the reverse: when reality disagrees with an item, fix the actual problem rather than working the checklist. This is a checkpoint, not a final answer: do not end your turn on the plan alone — after recording the todo list, continue the task; do not stop here. diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index b6abff059..24c2ca15a 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -109,7 +109,7 @@ import { obfuscateProviderContext, SecretObfuscator, } from "./secrets"; -import { AgentSession, type Downshift, type PlanYolo } from "./session/agent-session"; +import { AgentSession, type PlanYolo, type Prewalk } from "./session/agent-session"; import { discoverAuthStorage as discoverAuthStorageFromConfig } from "./session/auth-broker-config"; import type { AuthStorage } from "./session/auth-storage"; import { @@ -407,8 +407,8 @@ export interface CreateAgentSessionOptions { thinkingLevel?: ConfiguredThinkingLevel; /** Models available for cycling (Ctrl+P in interactive mode) */ scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; - /** Downshift from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ - downshift?: Downshift; + /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ + prewalk?: Prewalk; /** Force read-only plan mode at start, auto-approve on the model's first resolve call, then switch to execute. */ planYolo?: PlanYolo; @@ -1987,7 +1987,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } // Resolve deferred --model/subagent patterns now that extension models are - // registered. Expand role aliases (`pi/smol`) and comma chains to concrete + // registered. Expand role aliases (`@smol`) and comma chains to concrete // selectors first so deferred resolution accepts everything the immediate // path (resolveModelOverride → resolveModelRoleValue) accepts. if (!model && deferredModelPatterns.length > 0) { @@ -2867,7 +2867,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} agent, pruneToolDescriptions: inlineToolDescriptors, thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel, - downshift: options.downshift, + prewalk: options.prewalk, planYolo: options.planYolo, serviceTierByFamily: initialServiceTierByFamily, sessionManager, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 0a1ec8332..37764e1d9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -260,9 +260,6 @@ import goalModeContextPrompt from "../prompts/goals/goal-mode-context.md" with { import goalTodoContextPrompt from "../prompts/goals/goal-todo-context.md" with { type: "text" }; import parentIrcSteerTemplate from "../prompts/steering/parent-irc.md" with { type: "text" }; import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type: "text" }; -import downshiftChecklistPrompt from "../prompts/system/downshift-checklist.md" with { type: "text" }; -import downshiftContinuePrompt from "../prompts/system/downshift-continue.md" with { type: "text" }; -import downshiftPlanPrompt from "../prompts/system/downshift-plan.md" with { type: "text" }; import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; @@ -277,6 +274,9 @@ import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool type: "text", }; import planYoloHandoffPrompt from "../prompts/system/plan-yolo-handoff.md" with { type: "text" }; +import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; +import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; +import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; import rewindReportTemplate from "../prompts/system/rewind-report.md" with { type: "text" }; import sideChannelNoToolsReminder from "../prompts/system/side-channel-no-tools.md" with { type: "text" }; import thinkingLoopRedirectTemplate from "../prompts/system/thinking-loop-redirect.md" with { type: "text" }; @@ -420,29 +420,29 @@ const MID_RUN_TODO_NUDGE_MUTATING_TOOLS: Record = { /** `customType` for the hidden mid-run todo nudge; `display: false`, so it reaches * the model but never renders in the TUI or transcript. */ const MID_RUN_TODO_NUDGE_MESSAGE_TYPE = "mid-run-todo-nudge"; -/** Hidden plan nudge injected by downshift; scrubbed from the LLM context +/** Hidden plan nudge injected by prewalk; scrubbed from the LLM context * when the switch happens. */ -const DOWNSHIFT_PLAN_MESSAGE_TYPE = "downshift-plan"; +const PREWALK_PLAN_MESSAGE_TYPE = "prewalk-plan"; /** Hidden safety-net nudge forcing one more turn after a text-only reply to * the plan nudge, which would otherwise end the run with no code written. */ -const DOWNSHIFT_CONTINUE_MESSAGE_TYPE = "downshift-continue"; +const PREWALK_CONTINUE_MESSAGE_TYPE = "prewalk-continue"; /** Hidden "verify before finishing" checklist steered into the run at the * switch, aimed at the fast model's specific failure patterns: partial * multi-site fixes, unnecessarily broad rewrites, and reported-test-only * verification. */ -const DOWNSHIFT_CHECKLIST_MESSAGE_TYPE = "downshift-checklist"; +const PREWALK_CHECKLIST_MESSAGE_TYPE = "prewalk-checklist"; /** Tools whose first successful call triggers the switch — once the todo - * gate is open (see {@link AgentSession.#downshiftTodoSeen}). Bash is + * gate is open (see {@link AgentSession.#prewalkTodoSeen}). Bash is * deliberately excluded: it doubles as exploration (ls/cat) and fired * turn-1 switches in practice. `todo` is deliberately NOT a trigger: firing * at the todo init handed the fast model 100% of the implementation with * zero started work and measurably regressed pass rates. */ -const DOWNSHIFT_ACTION_TOOLS: Record = { +const PREWALK_ACTION_TOOLS: Record = { edit: true, write: true, }; /** `customType` for the hidden hand-off message steered to the target model - * once PlanYolo auto-approves the plan. Unlike downshift's plan nudge this + * once PlanYolo auto-approves the plan. Unlike prewalk's plan nudge this * is never scrubbed — it IS the instruction the target model acts on. */ const PLAN_YOLO_HANDOFF_MESSAGE_TYPE = "plan-yolo-handoff"; /** Abort reason for the Gemini reasoning-header runaway interrupt. Surfaced on the @@ -713,7 +713,7 @@ export interface AsyncJobSnapshot { export type { ShakeMode, ShakeResult }; /** - * Downshift: switches an active session one-way from its starting model to + * Prewalk: switches an active session one-way from its starting model to * a fast/cheap `target` at the first completed turn that runs an edit/write * tool once the todo list exists. A hidden plan nudge asks the starting * model to write a plan, initialize its todo list from it, and start; the @@ -723,7 +723,7 @@ export type { ShakeMode, ShakeResult }; * finishing. Both are always on — this is the one mechanism that won out * over turn-count and ungated variants in testing. */ -export interface Downshift { +export interface Prewalk { target: Model; thinkingLevel?: ConfiguredThinkingLevel; } @@ -755,8 +755,8 @@ export interface AgentSessionConfig { scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; /** Initial session thinking selector. */ thinkingLevel?: ConfiguredThinkingLevel; - /** Downshift from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ - downshift?: Downshift; + /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ + prewalk?: Prewalk; /** Force read-only plan mode at start, auto-approve on the model's first * `resolve` call, then switch to the target to implement. */ planYolo?: PlanYolo; @@ -1678,13 +1678,13 @@ export class AgentSession { #autoThinking: boolean = false; /** The level `auto` last resolved to (for UI); undefined until a turn is classified. */ #autoResolvedLevel: Effort | undefined; - #downshift: Downshift | undefined; + #prewalk: Prewalk | undefined; /** True once the plan nudge has been queued; scrubbed from context at the switch. */ - #downshiftPlanInjected = false; - /** True once any successful `todo` call landed — opens the downshift + #prewalkPlanInjected = false; + /** True once any successful `todo` call landed — opens the prewalk * trigger gate: the switch fires at the first edit/write AFTER the todo * list exists (sessions without a todo tool skip the gate). */ - #downshiftTodoSeen = false; + #prewalkTodoSeen = false; #planYolo: PlanYolo | undefined; #planYoloPreviousTools: string[] | undefined; #planYoloArmed = false; @@ -2178,10 +2178,10 @@ export class AgentSession { this.#emit(pending); } - /** Advance the one-way downshift switch at a completed assistant-turn boundary. */ - async #advanceDownshift(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { - const downshift = this.#downshift; - if (!downshift || context?.message.role !== "assistant") return; + /** Advance the one-way prewalk switch at a completed assistant-turn boundary. */ + async #advancePrewalk(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { + const prewalk = this.#prewalk; + if (!prewalk || context?.message.role !== "assistant") return; // Structural safety net: every branch below assumes the agent loop will // run another turn. It won't if THIS turn had no tool calls — the loop @@ -2191,11 +2191,11 @@ export class AgentSession { // silently killing production SWE-bench runs before any code was ever // written. Force one more turn only in that specific, self-created // hazard window. - if (this.#downshiftPlanInjected && context.toolResults.length === 0) { + if (this.#prewalkPlanInjected && context.toolResults.length === 0) { this.agent.steer({ role: "custom", - customType: DOWNSHIFT_CONTINUE_MESSAGE_TYPE, - content: downshiftContinuePrompt, + customType: PREWALK_CONTINUE_MESSAGE_TYPE, + content: prewalkContinuePrompt, attribution: "agent", display: false, timestamp: Date.now(), @@ -2209,24 +2209,24 @@ export class AgentSession { // the fast model the whole implementation cold. Sessions without a todo // tool skip the gate. if (context.toolResults.some(result => result.toolName === "todo")) { - this.#downshiftTodoSeen = true; + this.#prewalkTodoSeen = true; } - const todoGateOpen = this.#downshiftTodoSeen || !this.#toolRegistry.has("todo"); + const todoGateOpen = this.#prewalkTodoSeen || !this.#toolRegistry.has("todo"); const action = todoGateOpen - ? context.toolResults.find(result => DOWNSHIFT_ACTION_TOOLS[result.toolName]) + ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) : undefined; if (!action) { - if (!this.#downshiftPlanInjected) { - this.#downshiftPlanInjected = true; + if (!this.#prewalkPlanInjected) { + this.#prewalkPlanInjected = true; this.agent.steer({ role: "custom", - customType: DOWNSHIFT_PLAN_MESSAGE_TYPE, - content: downshiftPlanPrompt, + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, display: false, attribution: "agent", timestamp: Date.now(), }); - this.emitNotice("info", "Downshift: injected deep-plan nudge.", "downshift"); + this.emitNotice("info", "Prewalk: injected deep-plan nudge.", "prewalk"); } return; } @@ -2236,24 +2236,24 @@ export class AgentSession { await this.#waitForSessionMessagePersistence(toolResult); } - this.#scrubDownshiftPlanNudge(liveMessages); - const target = downshift.target; + this.#scrubPrewalkPlanNudge(liveMessages); + const target = prewalk.target; if (this.model && modelsAreEqual(this.model, target)) { - this.#downshift = undefined; + this.#prewalk = undefined; return; } - await this.setModelTemporary(target, downshift.thinkingLevel, { ephemeral: true }); - this.#downshift = undefined; + await this.setModelTemporary(target, prewalk.thinkingLevel, { ephemeral: true }); + this.#prewalk = undefined; this.emitNotice( "info", - `Downshift: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, - "downshift", + `Prewalk: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, + "prewalk", ); this.agent.steer({ role: "custom", - customType: DOWNSHIFT_CHECKLIST_MESSAGE_TYPE, - content: downshiftChecklistPrompt, + customType: PREWALK_CHECKLIST_MESSAGE_TYPE, + content: prewalkChecklistPrompt, attribution: "agent", display: false, timestamp: Date.now(), @@ -2261,35 +2261,35 @@ export class AgentSession { } /** - * Arm downshift outside the normal startup path (the `/downshift` slash + * Arm prewalk outside the normal startup path (the `/prewalk` slash * command): sets the target and immediately steers the plan nudge rather * than waiting for the next turn boundary, since an explicit manual - * invocation means "start this now." A no-op with a notice if a downshift + * invocation means "start this now." A no-op with a notice if a prewalk * is already armed and waiting. */ - armDownshift(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { - if (this.#downshift) { + armPrewalk(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { + if (this.#prewalk) { this.emitNotice( "info", - `Downshift: already armed for ${this.#downshift.target.provider}/${this.#downshift.target.id}, waiting for the first edit/write.`, - "downshift", + `Prewalk: already armed for ${this.#prewalk.target.provider}/${this.#prewalk.target.id}, waiting for the first edit/write.`, + "prewalk", ); return; } - this.#downshift = { target, thinkingLevel }; - this.#downshiftPlanInjected = true; + this.#prewalk = { target, thinkingLevel }; + this.#prewalkPlanInjected = true; this.agent.steer({ role: "custom", - customType: DOWNSHIFT_PLAN_MESSAGE_TYPE, - content: downshiftPlanPrompt, + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, display: false, attribution: "agent", timestamp: Date.now(), }); this.emitNotice( "info", - `Downshift: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, - "downshift", + `Prewalk: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, + "prewalk", ); } @@ -2299,12 +2299,12 @@ export class AgentSession { * Splices the loop's live context array in place (the run streams from * it) and mirrors the removal into agent state. The persisted transcript * keeps the message for audit; a session reload re-materializes it, - * which is acceptable for downshift's single-run lifecycle. + * which is acceptable for prewalk's single-run lifecycle. */ - #scrubDownshiftPlanNudge(liveMessages: AgentMessage[]): void { - if (!this.#downshiftPlanInjected) return; + #scrubPrewalkPlanNudge(liveMessages: AgentMessage[]): void { + if (!this.#prewalkPlanInjected) return; const isPlanNudge = (m: AgentMessage): boolean => - m.role === "custom" && m.customType === DOWNSHIFT_PLAN_MESSAGE_TYPE; + m.role === "custom" && m.customType === PREWALK_PLAN_MESSAGE_TYPE; for (let i = liveMessages.length - 1; i >= 0; i--) { if (isPlanNudge(liveMessages[i])) liveMessages.splice(i, 1); } @@ -2445,8 +2445,8 @@ export class AgentSession { } else { this.#thinkingLevel = config.thinkingLevel; } - if (config.downshift) { - this.#downshift = config.downshift; + if (config.prewalk) { + this.#prewalk = config.prewalk; } if (config.planYolo) { this.#planYolo = config.planYolo; @@ -2525,7 +2525,7 @@ export class AgentSession { }); if (detection) this.#maybeInjectToolCallLoopRedirect(messages, detection); } - await this.#advanceDownshift(messages, context); + await this.#advancePrewalk(messages, context); this.#advisorPrimaryTurnsCompleted++; if (this.#advisors.length > 0) { for (const a of this.#advisors) { @@ -7509,6 +7509,11 @@ export class AgentSession { return this.#planModeState; } + /** Prewalk state, if armed and active */ + getPrewalkState(): Prewalk | undefined { + return this.#prewalk; + } + setPlanModeState(state: PlanModeState | undefined): void { this.#planModeState = state; if (state?.enabled) { @@ -9685,7 +9690,7 @@ export class AgentSession { const all = this.#modelRegistry.getAvailable(); const patterns = this.settings.get("enabledModels"); if (!patterns || patterns.length === 0) return all; - return filterAvailableModelsByEnabledPatterns(all, patterns); + return filterAvailableModelsByEnabledPatterns(all, patterns, this.settings); } // ========================================================================= diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 72ec7cfb9..14843a6ee 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -462,11 +462,11 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ }, }, { - name: "downshift", - description: "Switch to a fast/cheap model at the next action (works even without --downshift)", - acpDescription: "Downshift at the next action", + name: "prewalk", + description: "Switch to a fast/cheap model at the next action (works even without --prewalk)", + acpDescription: "Prewalk at the next action", handle: async (_command, runtime) => { - const rolePattern = expandRoleAlias("pi/smol", runtime.settings); + const rolePattern = expandRoleAlias("@smol", runtime.settings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry: runtime.session.modelRegistry, @@ -478,9 +478,9 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ if (!runtime.session.modelRegistry.hasConfiguredAuth(resolved.model)) { return usage(`No API key for ${resolved.model.provider}/${resolved.model.id}`, runtime); } - runtime.session.armDownshift(resolved.model, resolved.thinkingLevel); + runtime.session.armPrewalk(resolved.model, resolved.thinkingLevel); await runtime.output( - `Downshift on: switching to ${resolved.model.provider}/${resolved.model.id} at the next edit/write (todo-gated).`, + `Prewalk on: switching to ${resolved.model.provider}/${resolved.model.id} at the next edit/write (todo-gated).`, ); return commandConsumed(); }, diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index caf8f46ad..2b071e4a4 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -50,7 +50,7 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ name: "task", description: "General-purpose subagent with full capabilities for delegated multi-step tasks", spawns: "*", - model: "pi/task", + model: "@task", thinkingLevel: AUTO_THINKING, }, template: taskMd, @@ -60,7 +60,7 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ frontmatter: { name: "sonic", description: "Low-reasoning agent for strictly mechanical updates or data collection only", - model: "pi/smol", + model: "@smol", thinkingLevel: Effort.Medium, }, template: taskMd, diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 0bf10760c..a872b93da 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -62,18 +62,25 @@ const THINKING_LEVEL_BY_SELECTOR: Readonly> = { }; function getOwnSelector(selectors: Readonly>, value: string | null | undefined): T | undefined { - return value === undefined || value === null || !Object.hasOwn(selectors, value) ? undefined : selectors[value]; + if (value === undefined || value === null) return undefined; + if (Object.hasOwn(selectors, value)) return selectors[value]; + // Accept unambiguous abbreviations (`xhi` → xhigh, `med` → medium) so every + // selector surface (`--thinking`, `:suffix`, role values) parses alike. + // Two-character minimum keeps single letters (`m`) from guessing. + if (value.length < 2) return undefined; + const matches = Object.keys(selectors).filter(selector => selector.startsWith(value)); + return matches.length === 1 ? selectors[matches[0]] : undefined; } /** - * Parses a provider-facing effort value. + * Parses a provider-facing effort value. Accepts unambiguous abbreviations. */ export function parseEffort(value: string | null | undefined): Effort | undefined { return getOwnSelector(EFFORT_BY_SELECTOR, value); } /** - * Parses an agent-local thinking selector. + * Parses an agent-local thinking selector. Accepts unambiguous abbreviations. */ export function parseThinkingLevel(value: string | null | undefined): ThinkingLevel | undefined { return getOwnSelector(THINKING_LEVEL_BY_SELECTOR, value); diff --git a/packages/coding-agent/src/tiny/models.ts b/packages/coding-agent/src/tiny/models.ts index 77c13426e..7500e2e25 100644 --- a/packages/coding-agent/src/tiny/models.ts +++ b/packages/coding-agent/src/tiny/models.ts @@ -1,4 +1,4 @@ -/** Default session-title model: the online pi/smol path (no local download / on-device inference). */ +/** Default session-title model: the online @smol path (no local download / on-device inference). */ export const ONLINE_TINY_TITLE_MODEL_KEY = "online"; /** Local model the `tiny-models` CLI downloads when none is named. Not the session-title default — that is {@link ONLINE_TINY_TITLE_MODEL_KEY}. */ export const DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY = "lfm2-700m"; @@ -87,9 +87,9 @@ void TINY_TITLE_MODEL_VALUES_MATCH_REGISTRY; export const TINY_TITLE_MODEL_OPTIONS = [ { value: ONLINE_TINY_TITLE_MODEL_KEY, - label: "Online (TINY role, else pi/smol)", + label: "Online (TINY role, else @smol)", description: - "Online title generation: the TINY model role (set one in /models) when assigned, otherwise the online fallback (commit role, then pi/smol). No local download or on-device inference.", + "Online title generation: the TINY model role (set one in /models) when assigned, otherwise the online fallback (commit role, then @smol). No local download or on-device inference.", }, ...TINY_TITLE_LOCAL_MODELS.map(model => ({ value: model.key, @@ -194,9 +194,9 @@ void TINY_MEMORY_MODEL_VALUES_MATCH_REGISTRY; export const TINY_MEMORY_MODEL_OPTIONS = [ { value: ONLINE_MEMORY_MODEL_KEY, - label: "Online (TINY role, else smol)", + label: "Online (TINY role, else @smol)", description: - "Use the online model: the TINY role from /models when set, otherwise pi/smol. No local model download or on-device inference.", + "Use the online model: the TINY role from /models when set, otherwise @smol. No local model download or on-device inference.", }, ...TINY_MEMORY_LOCAL_MODELS.map(model => ({ value: model.key, @@ -256,9 +256,9 @@ export type AutoThinkingModelKey = TinyMemoryModelKey; export const AUTO_THINKING_MODEL_OPTIONS = [ { value: ONLINE_AUTO_THINKING_MODEL_KEY, - label: "Online (TINY role, else smol)", + label: "Online (TINY role, else @smol)", description: - "Classify prompt difficulty online with the TINY role model (set one in /models) or pi/smol; no local download or on-device inference.", + "Classify prompt difficulty online with the TINY role model (set one in /models) or @smol; no local download or on-device inference.", }, ...TINY_MEMORY_LOCAL_MODELS.map(model => ({ value: model.key, diff --git a/packages/coding-agent/src/tools/bash-interactive.ts b/packages/coding-agent/src/tools/bash-interactive.ts index fc39cb0c8..8618146fe 100644 --- a/packages/coding-agent/src/tools/bash-interactive.ts +++ b/packages/coding-agent/src/tools/bash-interactive.ts @@ -19,6 +19,7 @@ import { OutputSink, type OutputSummary } from "../session/streaming-output"; import { sanitizeWithOptionalSixelPassthrough } from "../utils/sixel"; import { resolveOutputMaxColumns, resolveOutputSinkHeadBytes } from "./output-meta"; import { formatStatusIcon, replaceTabs } from "./render-utils"; +import { readTerminalRows, styleTerminalRow } from "./terminal-output"; export interface BashInteractiveResult extends OutputSummary { exitCode: number | undefined; @@ -225,14 +226,10 @@ class BashInteractiveOverlayComponent implements Component { #readViewport(innerWidth: number, maxContentRows: number): string[] { this.#terminal.resize(innerWidth, maxContentRows); - const buffer = this.#terminal.buffer.active; - const viewportY = buffer.viewportY; - const visibleLines: string[] = []; - for (let i = 0; i < maxContentRows; i++) { - const line = buffer.getLine(viewportY + i)?.translateToString(true) ?? ""; - visibleLines.push(truncateToWidth(replaceTabs(sanitizeText(line)), innerWidth)); - } - return visibleLines; + const viewportY = this.#terminal.buffer.active.viewportY; + return readTerminalRows(this.#terminal, viewportY, maxContentRows).map(line => + truncateToWidth(styleTerminalRow(line, this.uiTheme.getFgAnsi("toolOutput")), innerWidth), + ); } render(width: number): readonly string[] { const safeWidth = Math.max(20, width); diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index e272c230c..ccfea33c1 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -36,12 +36,18 @@ import { stripRawOutputArtifactNotice, } from "./output-meta"; import { resolveToCwd } from "./path-utils"; -import { capPreviewLines, formatToolWorkingDirectory, previewWindowRows, replaceTabs } from "./render-utils"; +import { + capPreviewLines, + DEFAULT_TERMINAL_PREVIEW_LINES, + formatToolWorkingDirectory, + previewWindowRows, + replaceTabs, +} from "./render-utils"; import { ToolAbortError, ToolError } from "./tool-errors"; import { toolResult } from "./tool-result"; import { clampTimeout, TOOL_TIMEOUTS } from "./tool-timeouts"; -export const BASH_DEFAULT_PREVIEW_LINES = 10; +export const BASH_DEFAULT_PREVIEW_LINES = DEFAULT_TERMINAL_PREVIEW_LINES; const BASH_ENV_NAME_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/; const DEFAULT_AUTO_BACKGROUND_THRESHOLD_MS = 60_000; diff --git a/packages/coding-agent/src/tools/browser/launch.ts b/packages/coding-agent/src/tools/browser/launch.ts index 38344bff7..51a45b850 100644 --- a/packages/coding-agent/src/tools/browser/launch.ts +++ b/packages/coding-agent/src/tools/browser/launch.ts @@ -121,12 +121,16 @@ async function loadBrowsers(): Promise { } /** - * Lazily download Chromium on first browser launch via @puppeteer/browsers. - * Skipped when a system Chromium (NixOS) or PUPPETEER_EXECUTABLE_PATH is set. - * The browser is cached under ~/.omp/puppeteer (getPuppeteerDir). + * Resolve the Chromium executable puppeteer will launch, lazily downloading it + * on first use via @puppeteer/browsers. Skipped when a system Chromium (NixOS) + * or PUPPETEER_EXECUTABLE_PATH is set. The browser is cached under + * ~/.omp/puppeteer (getPuppeteerDir). Returns undefined when platform + * detection fails (puppeteer default resolution takes over). Exported so + * real-browser tests can probe launchability and skip on hosts missing + * Chrome's system libraries. */ let chromiumExecutablePromise: Promise | undefined; -async function ensureChromiumExecutable(): Promise { +export async function ensureChromiumExecutable(): Promise { const sysChrome = resolveSystemChromium(); if (sysChrome) return sysChrome; const envPath = process.env.PUPPETEER_EXECUTABLE_PATH; diff --git a/packages/coding-agent/src/tools/inspect-image.ts b/packages/coding-agent/src/tools/inspect-image.ts index a270c5dbc..f090225f1 100644 --- a/packages/coding-agent/src/tools/inspect-image.ts +++ b/packages/coding-agent/src/tools/inspect-image.ts @@ -157,8 +157,8 @@ export class InspectImageTool implements AgentTool `- ${daemonLabel(daemon)}`).join("\n") : "No daemons."; - case "logs": - return `${result.text}${result.text && !result.text.endsWith("\n") ? "\n" : ""}[${result.name}: ${result.state}; cursor=${result.cursor}${result.timedOut ? "; follow timed out" : ""}]`; + case "logs": { + const text = sanitizeText(result.text); + return `${text}${text && !text.endsWith("\n") ? "\n" : ""}[${result.name}: ${result.state}; cursor=${result.cursor}${result.timedOut ? "; follow timed out" : ""}]`; + } case "wait": { const lines = [daemonLabel(result.daemon)]; if (result.matched) lines.push(`Matched: ${result.matched}`); @@ -270,14 +277,28 @@ function toolContent(result: DaemonRpcResult, params: LaunchParams): string { } } -function toolDetails(result: DaemonRpcResult): LaunchToolDetails { +async function toolDetails(result: DaemonRpcResult, params: LaunchParams): Promise { switch (result.op) { case "start": return { op: "start", daemon: result.daemon, timedOut: result.readyTimedOut }; case "list": return { op: "list", daemons: result.daemons }; - case "logs": - return { op: "logs", cursor: result.cursor, timedOut: result.timedOut, state: result.state }; + case "logs": { + const terminalRows = + result.terminalText === undefined + ? undefined + : await renderTerminalOutput(result.terminalText, { + head: params.head ?? false, + maxRows: Math.min(1_000, Math.floor(params.lines ?? 100)), + }); + return { + op: "logs", + cursor: result.cursor, + timedOut: result.timedOut, + state: result.state, + terminalRows, + }; + } case "wait": return { op: "wait", daemon: result.daemon, timedOut: result.timedOut, matched: result.matched }; case "send": @@ -372,7 +393,7 @@ export class LaunchTool implements AgentTool { + const innerWidth = outputBlockContentWidth(width); + const rows = body.map(line => truncateToWidth(line, innerWidth)); + return { + header, + state: options.isPartial ? "pending" : failed ? "error" : "success", + sections: [ + { + label: theme.fg("toolTitle", "Output"), + lines: capPreviewLines(rows, theme, { + expanded: options.expanded, + max: DEFAULT_TERMINAL_PREVIEW_LINES, + }), + }, + ], + width, + }; + }); + } + return createCachedComponent( () => options.expanded, (width, expanded) => { let visible = body; - if (op === "logs") { - visible = capPreviewLines(body, theme, { expanded }); - } else if (!expanded && op === "list" && body.length > PREVIEW_LIMITS.COLLAPSED_ITEMS) { + if (!expanded && op === "list" && body.length > PREVIEW_LIMITS.COLLAPSED_ITEMS) { const remaining = body.length - PREVIEW_LIMITS.COLLAPSED_ITEMS; visible = [ ...body.slice(0, PREVIEW_LIMITS.COLLAPSED_ITEMS), diff --git a/packages/coding-agent/src/tools/render-utils.ts b/packages/coding-agent/src/tools/render-utils.ts index 12228a1cb..9a6e857af 100644 --- a/packages/coding-agent/src/tools/render-utils.ts +++ b/packages/coding-agent/src/tools/render-utils.ts @@ -62,6 +62,9 @@ export const PREVIEW_LIMITS = { DIFF_COLLAPSED_LINES: 40, } as const; +/** Default number of terminal output rows shown before expansion. */ +export const DEFAULT_TERMINAL_PREVIEW_LINES = 10; + /** Truncation lengths for different content types */ export const TRUNCATE_LENGTHS = { /** Short titles, labels */ diff --git a/packages/coding-agent/src/tools/terminal-output.ts b/packages/coding-agent/src/tools/terminal-output.ts new file mode 100644 index 000000000..6f1043ff4 --- /dev/null +++ b/packages/coding-agent/src/tools/terminal-output.ts @@ -0,0 +1,141 @@ +import { sanitizeText } from "@oh-my-pi/pi-utils"; +import type { Terminal as XtermTerminal } from "@xterm/headless"; + +const RESET = "\x1b[0m"; +const SGR = /\x1b\[([0-9;]*)m/g; + +interface TerminalCell { + getChars(): string; + getWidth(): number; + getFgColor(): number; + getBgColor(): number; + isBold(): number; + isDim(): number; + isItalic(): number; + isUnderline(): number; + isInverse(): number; + isStrikethrough(): number; + isOverline(): number; + isFgRGB(): boolean; + isBgRGB(): boolean; + isFgPalette(): boolean; + isBgPalette(): boolean; +} + +function addColor(codes: number[], cell: TerminalCell, foreground: boolean): void { + const rgb = foreground ? cell.isFgRGB() : cell.isBgRGB(); + const palette = foreground ? cell.isFgPalette() : cell.isBgPalette(); + if (!rgb && !palette) return; + + const color = foreground ? cell.getFgColor() : cell.getBgColor(); + codes.push(foreground ? 38 : 48); + if (rgb) { + codes.push(2, (color >> 16) & 0xff, (color >> 8) & 0xff, color & 0xff); + } else { + codes.push(5, color); + } +} + +function cellStyle(cell: TerminalCell): string { + const codes: number[] = []; + if (cell.isBold() !== 0) codes.push(1); + if (cell.isDim() !== 0) codes.push(2); + if (cell.isItalic() !== 0) codes.push(3); + if (cell.isUnderline() !== 0) codes.push(4); + if (cell.isInverse() !== 0) codes.push(7); + if (cell.isStrikethrough() !== 0) codes.push(9); + if (cell.isOverline() !== 0) codes.push(53); + addColor(codes, cell, true); + addColor(codes, cell, false); + return codes.length > 0 ? `\x1b[${codes.join(";")}m` : ""; +} + +function isSafeStyle(codes: readonly number[]): boolean { + let index = 0; + while (index < codes.length) { + const code = codes[index++]; + if (code === 1 || code === 2 || code === 3 || code === 4 || code === 7 || code === 9 || code === 53) continue; + if (code !== 38 && code !== 48) return false; + const mode = codes[index++]; + if (mode === 5) { + const color = codes[index++]; + if (color === undefined || color < 0 || color > 255) return false; + continue; + } + if (mode !== 2) return false; + for (let channel = 0; channel < 3; channel++) { + const color = codes[index++]; + if (color === undefined || color < 0 || color > 255) return false; + } + } + return true; +} + +/** Applies the active tool-output color while preserving safe styles from a virtual terminal row. */ +export function styleTerminalRow(row: string, baseForeground: string): string { + let output = baseForeground; + let offset = 0; + let hasText = false; + for (const match of row.matchAll(SGR)) { + const index = match.index ?? 0; + const text = sanitizeText(row.slice(offset, index)); + output += text; + hasText ||= text.length > 0; + + const codes = match[1].split(";").map(Number); + if (match[1] === "0") output += `${RESET}${baseForeground}`; + else if (codes.length > 0 && codes.every(Number.isInteger) && isSafeStyle(codes)) output += match[0]; + offset = index + match[0].length; + } + const text = sanitizeText(row.slice(offset)); + output += text; + hasText ||= text.length > 0; + return hasText ? `${output}${RESET}` : ""; +} + +/** Reads terminal screen rows as sanitized text plus only the styles the TUI may replay. */ +export function readTerminalRows(terminal: XtermTerminal, startRow: number, rowCount: number): string[] { + const buffer = terminal.buffer.active; + const reusableCell = buffer.getNullCell(); + const rows: string[] = []; + const endRow = Math.min(buffer.length, Math.max(0, startRow) + Math.max(0, rowCount)); + + for (let row = Math.max(0, startRow); row < endRow; row++) { + const line = buffer.getLine(row); + if (!line) { + rows.push(""); + continue; + } + + const cells: Array<{ chars: string; style: string }> = []; + let lastContent = -1; + for (let column = 0; column < line.length; ) { + const cell = line.getCell(column, reusableCell); + if (!cell) break; + const chars = cell.getChars(); + const width = Math.max(1, cell.getWidth()); + cells.push({ chars: chars || " ", style: cellStyle(cell) }); + if (chars && chars !== " ") lastContent = cells.length - 1; + column += width; + } + + if (lastContent < 0) { + rows.push(""); + continue; + } + + let rendered = ""; + let previousStyle: string | undefined; + for (let index = 0; index <= lastContent; index++) { + const cell = cells[index]!; + if (cell.style !== previousStyle) { + rendered += `${RESET}${cell.style}`; + previousStyle = cell.style; + } + rendered += cell.chars; + } + rows.push(rendered); + } + + return rows; +} diff --git a/packages/coding-agent/src/tts/speech-enhancer.ts b/packages/coding-agent/src/tts/speech-enhancer.ts index f53f2d188..8f100d63f 100644 --- a/packages/coding-agent/src/tts/speech-enhancer.ts +++ b/packages/coding-agent/src/tts/speech-enhancer.ts @@ -70,10 +70,10 @@ export class SpeechEnhancer { async rewrite(block: string, signal?: AbortSignal): Promise { try { const { settings, registry, sessionId } = this.#deps; - // `pi/tiny` expands a configured `modelRoles.tiny` and otherwise falls + // `@tiny` expands a configured `modelRoles.tiny` and otherwise falls // through tiny's alias to the smol priority chain — unlike bare role // lookup, this resolves even with no roles configured. - const model = resolveModelRoleValue("pi/tiny", registry.getAvailable(), { + const model = resolveModelRoleValue("@tiny", registry.getAvailable(), { settings, matchPreferences: getModelMatchPreferences(settings), }).model; diff --git a/packages/coding-agent/src/utils/image-vision-fallback.ts b/packages/coding-agent/src/utils/image-vision-fallback.ts index bf89a903d..26d34f032 100644 --- a/packages/coding-agent/src/utils/image-vision-fallback.ts +++ b/packages/coding-agent/src/utils/image-vision-fallback.ts @@ -97,7 +97,7 @@ function formatImageBlock(localUrl: string, description: string): string { /** * Resolve a vision-capable model, mirroring the inspect_image priority - * (`pi/vision` → `pi/default` → active → first image-capable available), but + * (`@vision` → `@default` → active → first image-capable available), but * never returning a text-only model. */ function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model | undefined { @@ -111,8 +111,8 @@ function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model | unde return model?.input.includes("image") ? model : undefined; }; return ( - resolvePattern("pi/vision") ?? - resolvePattern("pi/default") ?? + resolvePattern("@vision") ?? + resolvePattern("@default") ?? resolvePattern(deps.activeModelString) ?? available.find(model => model.input.includes("image")) ); diff --git a/packages/coding-agent/src/vibe/runtime.ts b/packages/coding-agent/src/vibe/runtime.ts index d7e1215aa..7d55b48c5 100644 --- a/packages/coding-agent/src/vibe/runtime.ts +++ b/packages/coding-agent/src/vibe/runtime.ts @@ -39,8 +39,8 @@ export type VibeCli = "fast" | "good"; /** * CLI flavor → bundled agent type. This IS the model-tier mapping: `sonic` - * carries `model: "pi/smol"` (the configured fast/low-latency role) and `task` - * carries `model: "pi/task"` (inherits the session's strong model). + * carries `model: "@smol"` (the configured fast/low-latency role) and `task` + * carries `model: "@task"` (inherits the session's strong model). * Resolution goes through {@link resolveAgentModelPatterns} exactly like a * `task` spawn, so `task.agentModelOverrides` and model-role settings apply. */ diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index f8892c557..886fedb6c 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -2244,6 +2244,13 @@ describe("ACP agent", () => { return { connection, calls }; } + /** Narrows `CreateElicitationRequest` to the `mode: "form"` branch; the SDK's `mode: string` catch-all arm otherwise defeats literal narrowing on `mode !== "form"`. */ + function isFormElicitation( + request: CreateElicitationRequest, + ): request is Extract { + return request.mode === "form"; + } + it("translates select to a single-property string-enum elicitation", async () => { const { connection, calls } = createElicitConnection(async () => ({ action: "accept", @@ -2258,7 +2265,7 @@ describe("ACP agent", () => { const request = calls[0]!; expect(request.mode).toBe("form"); expect(request.message).toBe("Pick one"); - if (request.mode !== "form" || !("sessionId" in request)) { + if (!isFormElicitation(request) || !("sessionId" in request)) { throw new Error("expected session-scoped form elicitation"); } expect(request.sessionId).toBe("session-select"); @@ -2281,7 +2288,7 @@ describe("ACP agent", () => { expect(result).toBe(true); expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") { + if (!isFormElicitation(request)) { throw new Error("expected form-mode elicitation"); } expect(request.message).toBe("Proceed?\n\nThis will overwrite the file."); @@ -2301,7 +2308,7 @@ describe("ACP agent", () => { expect(result).toBe("claude"); expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") { + if (!isFormElicitation(request)) { throw new Error("expected form-mode elicitation"); } expect(request.message).toBe("Your name?"); @@ -2450,7 +2457,7 @@ describe("ACP agent", () => { expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") throw new Error("expected form-mode elicitation"); + if (!isFormElicitation(request)) throw new Error("expected form-mode elicitation"); expect(request.requestedSchema.properties?.value).toEqual({ type: "string" }); }); diff --git a/packages/coding-agent/test/agent-session-downshift.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts similarity index 94% rename from packages/coding-agent/test/agent-session-downshift.test.ts rename to packages/coding-agent/test/agent-session-prewalk.test.ts index 9b697ba26..94a903870 100644 --- a/packages/coding-agent/test/agent-session-downshift.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -13,21 +13,21 @@ import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manage import { TempDir } from "@oh-my-pi/pi-utils"; /** - * Downshift: one-way switch from the starting model to a fast/cheap target + * Prewalk: one-way switch from the starting model to a fast/cheap target * at the first completed turn that starts execution — an edit/write tool, * or the todo-list init the plan nudge asks for — with a hidden plan nudge * before the switch and a hidden verify-before-finishing checklist after * it. This is the single mechanism that won out over fixed-turn and * ungated variants in benchmark testing — see the plan nudge / checklist / - * continuation-safety-net prompts under `src/prompts/system/downshift-*.md`. + * continuation-safety-net prompts under `src/prompts/system/prewalk-*.md`. */ -describe("AgentSession downshift", () => { +describe("AgentSession prewalk", () => { let tempDir: TempDir; let authStorage: AuthStorage; let session: AgentSession | undefined; beforeEach(async () => { - tempDir = TempDir.createSync("@pi-downshift-"); + tempDir = TempDir.createSync("@pi-prewalk-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db")); authStorage.setRuntimeApiKey("anthropic", "test-key"); }); @@ -110,7 +110,7 @@ describe("AgentSession downshift", () => { }); } - it("downshifts at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => { + it("prewalks at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); @@ -155,7 +155,7 @@ describe("AgentSession downshift", () => { settings: Settings.isolated({ "compaction.enabled": false }), modelRegistry, toolRegistry, - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -217,7 +217,7 @@ describe("AgentSession downshift", () => { settings: Settings.isolated({ "compaction.enabled": false }), modelRegistry, toolRegistry, - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -277,7 +277,7 @@ describe("AgentSession downshift", () => { [recordTool.name, recordTool as AgentTool], [writeTool.name, writeTool as AgentTool], ]), - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -292,13 +292,13 @@ describe("AgentSession downshift", () => { expect(session.model?.id).toBe(target.id); }); - it("armDownshift (the /downshift slash command) pre-arms the switch for the very next edit/write", async () => { + it("armPrewalk (the /prewalk slash command) pre-arms the switch for the very next edit/write", async () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - // No `downshift` in the session config — this simulates a session that - // was NOT started with --downshift, forced on via the slash command. + // No `prewalk` in the session config — this simulates a session that + // was NOT started with --prewalk, forced on via the slash command. const mock = createMockModel({ responses: [toolCall("t1", "write"), { content: ["done"] }] }); const requested: string[] = []; const agent = new Agent({ @@ -325,8 +325,8 @@ describe("AgentSession downshift", () => { }); // Arming twice back-to-back must stay a single, idempotent arm. - session.armDownshift(target); - session.armDownshift(target); + session.armPrewalk(target); + session.armPrewalk(target); await session.prompt("do the task"); diff --git a/packages/coding-agent/test/agent-session-prune-persistence.test.ts b/packages/coding-agent/test/agent-session-prune-persistence.test.ts index 3b78c3657..438f0de4e 100644 --- a/packages/coding-agent/test/agent-session-prune-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-prune-persistence.test.ts @@ -114,7 +114,7 @@ describe("AgentSession per-turn prune persistence", () => { const message = session.agent.state.messages.find( candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID, ); - if (!message || message.role !== "toolResult" || !Array.isArray(message.content)) { + if (message?.role !== "toolResult" || !Array.isArray(message.content)) { throw new Error("Expected the seeded tool result in live agent state"); } const text = message.content.find(block => block.type === "text"); @@ -156,7 +156,7 @@ describe("AgentSession per-turn prune persistence", () => { const rebuilt = reloaded .buildSessionContext() .messages.find(candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID); - if (!rebuilt || rebuilt.role !== "toolResult" || !Array.isArray(rebuilt.content)) { + if (rebuilt?.role !== "toolResult" || !Array.isArray(rebuilt.content)) { throw new Error("Expected the seeded tool result in the from-disk rebuild"); } const rebuiltText = rebuilt.content.find(block => block.type === "text"); diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts index 0caf26767..8d206ca74 100644 --- a/packages/coding-agent/test/bundled-agent-parsing.test.ts +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -12,7 +12,7 @@ describe("bundled agent parsing", () => { expect(reviewer).toBeDefined(); expect(reviewer?.source).toBe("bundled"); - expect(reviewer?.model).toEqual(["pi/slow"]); + expect(reviewer?.model).toEqual(["@slow"]); expect(reviewer?.thinkingLevel).toBeUndefined(); }); @@ -20,7 +20,7 @@ describe("bundled agent parsing", () => { const task = getBundledAgent("task"); expect(task).toBeDefined(); - expect(task?.model).toEqual(["pi/task"]); + expect(task?.model).toEqual(["@task"]); expect(task?.thinkingLevel).toBe(AUTO_THINKING); }); diff --git a/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts b/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts index 4c315794c..14ae7530c 100644 --- a/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts +++ b/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts @@ -34,7 +34,7 @@ describe("commit role thinking selection", () => { const settings = createSettings({ default: `${defaultModel.provider}/${defaultModel.id}:high`, commit: `${commitModel.provider}/${commitModel.id}:low`, - smol: "pi/default:minimal", + smol: "@default:minimal", }); const registry = { getAvailable: () => [defaultModel, commitModel], diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 3e41f52a7..8265640bf 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -369,7 +369,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(finalLines).toContain("L8"); const text = result.content[0]?.type === "text" ? result.content[0].text : ""; - expect(text).toMatch(/Recovered from a stale file hash using a previous read snapshot/); + expect(text).toMatch(/Recovered by remapping stale line anchors to unchanged current lines/); }); }); @@ -437,7 +437,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(finalLines).toContain("GAMMA"); expect(finalLines).not.toContain("gamma"); const text = result.content[0]?.type === "text" ? result.content[0].text : ""; - expect(text).toMatch(/Recovered from a stale file hash using a previous read snapshot/); + expect(text).toMatch(/Recovered by remapping stale line anchors to unchanged current lines/); }); }); diff --git a/packages/coding-agent/test/extensibility/ext-model-query.test.ts b/packages/coding-agent/test/extensibility/ext-model-query.test.ts index db119ed73..c1d7c3469 100644 --- a/packages/coding-agent/test/extensibility/ext-model-query.test.ts +++ b/packages/coding-agent/test/extensibility/ext-model-query.test.ts @@ -60,7 +60,7 @@ describe("createExtensionModelQuery", () => { getModelRole: (role: string) => (role === "slow" ? "anthropic/claude-opus-4-8" : undefined), } as unknown as Settings; const q = createExtensionModelQuery(registry(), settings, () => undefined); - expect(q.resolve("pi/slow")).toBe(claude); + expect(q.resolve("@slow")).toBe(claude); }); test("family() groups a vendor's point releases and separates vendors", () => { diff --git a/packages/coding-agent/test/interactive-mode-status.test.ts b/packages/coding-agent/test/interactive-mode-status.test.ts index 2a2bff09c..542564c52 100644 --- a/packages/coding-agent/test/interactive-mode-status.test.ts +++ b/packages/coding-agent/test/interactive-mode-status.test.ts @@ -1,5 +1,6 @@ import { beforeAll, describe, expect, test, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; import { UiHelpers } from "@oh-my-pi/pi-coding-agent/modes/utils/ui-helpers"; @@ -64,7 +65,10 @@ function createInitialRenderHarness(): { ctx: InteractiveModeContext; helpers: U describe("InteractiveMode.showStatus", () => { beforeAll(async () => { - // showStatus uses the global theme instance + // showStatus uses the global theme instance; renderInitialMessages reads + // the global Settings (display.collapseCompacted). + resetSettingsForTest(); + await Settings.init({ inMemory: true }); await initTheme(); }); diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index 4e2e5d4fc..5b3152e40 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -17,6 +17,7 @@ import { resolveModelRoleValue, resolveModelScope, } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { DEFAULT_MODEL_ROLE_ALIAS, LEGACY_MODEL_ROLE_ALIAS_PREFIX } from "@oh-my-pi/pi-coding-agent/config/model-roles"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; // Mock models for testing @@ -623,12 +624,12 @@ describe("parseModelPattern", () => { }); describe("resolveModelRoleValue", () => { - test("resolves pi/: by expanding role alias before parsing thinking", () => { + test("resolves @role: by expanding role alias before parsing thinking", () => { const settings = { getModelRole: (role: string) => (role === "smol" ? "openrouter/qwen/qwen3-coder:exacto" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/smol:high", allModels, { settings }); + const result = resolveModelRoleValue("@smol:high", allModels, { settings }); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("qwen/qwen3-coder:exacto"); @@ -636,12 +637,12 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); - test("resolves pi/:max by expanding role alias before parsing thinking", () => { + test("resolves @role:max by expanding role alias before parsing thinking", () => { const settings = { getModelRole: (role: string) => (role === "smol" ? "openai-codex/gpt-5.3-codex" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/smol:max", allModels, { settings }); + const result = resolveModelRoleValue("@smol:max", allModels, { settings }); expect(result.model?.provider).toBe("openai-codex"); expect(result.model?.id).toBe("gpt-5.3-codex"); @@ -650,12 +651,12 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); - test("resolves pi/default through configured default role alias", () => { + test("resolves @default through configured default role alias", () => { const settings = { getModelRole: (role: string) => (role === "default" ? "openrouter/qwen/qwen3-coder:exacto" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/default", allModels, { settings }); + const result = resolveModelRoleValue("@default", allModels, { settings }); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("qwen/qwen3-coder:exacto"); @@ -736,13 +737,13 @@ describe("resolveModelRoleValue", () => { }); }); describe("resolveAgentModelPatterns", () => { - test("falls back to the active session model when pi/task is unset", () => { + test("falls back to the active session model when @task is unset", () => { const settings = Settings.isolated({ modelRoles: { default: "anthropic/claude-sonnet-4-5" }, }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, activeModelPattern: "openai/gpt-4o", }); @@ -759,7 +760,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, activeModelPattern: "openai/gpt-4o", }); @@ -775,7 +776,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, }); @@ -787,17 +788,17 @@ describe("resolveAgentModelPatterns", () => { modelRoles: { default: "local/llama" }, }); - expect(resolveAgentModelPatterns({ agentModel: "pi/smol", settings })).toEqual(["local/llama"]); - expect(resolveAgentModelPatterns({ agentModel: "pi/slow", settings })).toEqual(["local/llama"]); - expect(resolveAgentModelPatterns({ agentModel: "pi/designer", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@smol", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@slow", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@designer", settings })).toEqual(["local/llama"]); }); test("expands cross-role default aliases when inheriting for an unset role", () => { const settings = Settings.isolated({ - modelRoles: { default: "pi/slow", slow: "anthropic/claude-sonnet-4-5" }, + modelRoles: { default: "@slow", slow: "anthropic/claude-sonnet-4-5" }, }); - expect(resolveAgentModelPatterns({ agentModel: "pi/smol", settings })).toEqual(["anthropic/claude-sonnet-4-5"]); + expect(resolveAgentModelPatterns({ agentModel: "@smol", settings })).toEqual(["anthropic/claude-sonnet-4-5"]); }); test("prefers configured designer role override over priority defaults", () => { @@ -809,7 +810,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/designer", + agentModel: "@designer", settings, }); @@ -818,7 +819,7 @@ describe("resolveAgentModelPatterns", () => { test("slow priority falls forward to Opus 4.8 before older Opus aliases", () => { const settings = Settings.isolated(); - const patterns = resolveAgentModelPatterns({ agentModel: "pi/slow", settings }); + const patterns = resolveAgentModelPatterns({ agentModel: "@slow", settings }); const dottedRegistry = { getAvailable: () => [ @@ -898,6 +899,70 @@ describe("resolveCliModel", () => { expect(result.model?.id).toBe("gpt-4o"); }); + test("resolves configured custom, legacy, and default role aliases from --model", () => { + const registry = { + getAll: () => allModels, + }; + const settings = Settings.isolated({ + modelRoles: { + default: "openai/gpt-4o", + fable: "anthropic/claude-sonnet-4-5:high", + }, + }); + + const canonical = resolveCliModel({ + cliModel: "@fable", + modelRegistry: registry, + settings, + }); + const legacy = resolveCliModel({ + cliModel: `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}fable`, + modelRegistry: registry, + settings, + }); + const defaultRole = resolveCliModel({ + cliModel: DEFAULT_MODEL_ROLE_ALIAS, + modelRegistry: registry, + settings, + }); + + expect(canonical.error).toBeUndefined(); + expect(canonical.model?.provider).toBe("anthropic"); + expect(canonical.model?.id).toBe("claude-sonnet-4-5"); + expect(canonical.thinkingLevel).toBe(Effort.High); + expect(legacy).toEqual(canonical); + expect(defaultRole.model?.provider).toBe("openai"); + expect(defaultRole.model?.id).toBe("gpt-4o"); + }); + + test("splits thinking suffixes and abbreviations off the * default alias", () => { + const registry = { + getAll: () => allModels, + }; + const settings = Settings.isolated({ + modelRoles: { default: "anthropic/claude-sonnet-4-5" }, + }); + + const explicit = resolveCliModel({ + cliModel: `${DEFAULT_MODEL_ROLE_ALIAS}:high`, + modelRegistry: registry, + settings, + }); + const abbreviated = resolveCliModel({ + cliModel: `${DEFAULT_MODEL_ROLE_ALIAS}:xhi`, + modelRegistry: registry, + settings, + }); + + expect(explicit.error).toBeUndefined(); + expect(explicit.model?.id).toBe("claude-sonnet-4-5"); + expect(explicit.thinkingLevel).toBe(Effort.High); + // `xhi` → xhigh via unique-prefix parsing, then clamped to the model ladder. + expect(abbreviated.error).toBeUndefined(); + expect(abbreviated.model?.id).toBe("claude-sonnet-4-5"); + expect(abbreviated.thinkingLevel).toBe(Effort.High); + }); + test("resolves fuzzy patterns within an explicit provider", () => { const registry = { getAll: () => allModels, @@ -1108,6 +1173,25 @@ describe("resolveModelScope", () => { expect(scoped[0].model.id).toBe("gpt-5.5"); }); + test("resolves role aliases in --models scope to the role's model with its thinking level", async () => { + const settings = Settings.isolated({ + modelRoles: { fable: "anthropic/claude-sonnet-4-5:high" }, + }); + + const scoped = await resolveModelScope( + ["@fable", "openai/gpt-4o"], + { getAvailable: () => allModels }, + undefined, + settings, + ); + + expect(scoped).toHaveLength(2); + expect(scoped[0].model.id).toBe("claude-sonnet-4-5"); + expect(scoped[0].thinkingLevel).toBe(Effort.High); + expect(scoped[0].explicitThinkingLevel).toBe(true); + expect(scoped[1].model.id).toBe("gpt-4o"); + }); + test("applies max thinking selectors to glob scopes when no literal max ids match", async () => { const registry = { getAvailable: () => mockCodexOverlapModels, @@ -1272,18 +1356,18 @@ describe("resolveModelFromString", () => { }); describe("expandRoleAlias", () => { - test("expands pi/vision to configured vision role", () => { + test("expands @vision to configured vision role", () => { const settings = Settings.isolated(); settings.setModelRole("vision", "openai/gpt-4o"); - expect(expandRoleAlias("pi/vision", settings)).toBe("openai/gpt-4o"); + expect(expandRoleAlias("@vision", settings)).toBe("openai/gpt-4o"); }); - test("keeps pi/vision alias when vision role is unset", () => { + test("keeps @vision alias when vision role is unset", () => { const settings = Settings.isolated(); settings.setModelRole("default", "anthropic/claude-sonnet-4-5"); - expect(expandRoleAlias("pi/vision", settings)).toBe("pi/vision"); + expect(expandRoleAlias("@vision", settings)).toBe("@vision"); }); }); @@ -1305,7 +1389,7 @@ describe("extractExplicitThinkingSelector", () => { test("treats max on pi role aliases as an explicit selector before expansion", () => { const settings = Settings.isolated(); settings.setModelRole("smol", "nanogpt/coding-router:max"); - const result = extractExplicitThinkingSelector("pi/smol:max", settings, { + const result = extractExplicitThinkingSelector("@smol:max", settings, { isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); expect(result).toBe(Effort.Max); @@ -1442,6 +1526,15 @@ describe("filterAvailableModelsByEnabledPatterns", () => { expect(filterAvailableModelsByEnabledPatterns(models, [])).toEqual(models); }); + test("resolves role aliases to the role's model when settings are provided", () => { + const settings = Settings.isolated({ + modelRoles: { fable: "anthropic/claude-sonnet-4-5:high" }, + }); + const result = filterAvailableModelsByEnabledPatterns(models, ["@fable"], settings); + expect(result).toHaveLength(1); + expect(result[0].id).toBe("claude-sonnet-4-5"); + }); + test("filters by exact provider/modelId", () => { const result = filterAvailableModelsByEnabledPatterns(models, ["anthropic/claude-sonnet-4-5"]); expect(result).toHaveLength(1); diff --git a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts index 03f64ac3d..bc3df339f 100644 --- a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts +++ b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts @@ -11,7 +11,7 @@ * scrollback-clearing repaint (`clearTerminalHistory`). */ -import { afterEach, beforeAll, describe, expect, it, type Mock, vi } from "bun:test"; +import { afterEach, beforeAll, beforeEach, describe, expect, it, type Mock, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, ImageContent, Usage } from "@oh-my-pi/pi-ai"; import { kStreamingPartialJson } from "@oh-my-pi/pi-ai/utils/block-symbols"; @@ -28,6 +28,13 @@ beforeAll(() => { initTheme(); }); +beforeEach(async () => { + // afterEach resets Settings, but renderInitialMessages reads the global + // Settings (display.collapseCompacted) — re-init before every test. + resetSettingsForTest(); + await Settings.init({ inMemory: true }); +}); + const originalImageProtocol = TERMINAL.imageProtocol; afterEach(() => { diff --git a/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts b/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts index 239790ed8..fbc75a45b 100644 --- a/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts +++ b/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeAll, describe, expect, test, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ExtensionActions, ExtensionCommandContextActions, @@ -31,6 +32,9 @@ import { Container } from "@oh-my-pi/pi-tui"; * leaving two identical custom-message components in the chat. */ beforeAll(async () => { + // renderInitialMessages reads the global Settings (display.collapseCompacted). + resetSettingsForTest(); + await Settings.init({ inMemory: true }); await initTheme(); }); diff --git a/packages/coding-agent/test/role-thinking-helper-propagation.test.ts b/packages/coding-agent/test/role-thinking-helper-propagation.test.ts index 882d20644..9310e2833 100644 --- a/packages/coding-agent/test/role-thinking-helper-propagation.test.ts +++ b/packages/coding-agent/test/role-thinking-helper-propagation.test.ts @@ -39,7 +39,7 @@ describe("role thinking helper propagation", () => { const model = getModelOrThrow("claude-sonnet-4-5"); const settings = createSettings({ default: `${model.provider}/${model.id}:high`, - smol: "pi/default:minimal", + smol: "@default:minimal", }); const registry = { getAvailable: () => [model], @@ -85,7 +85,7 @@ describe("role thinking helper propagation", () => { const model = getModelOrThrow("claude-sonnet-4-5"); const settings = createSettings({ default: `${model.provider}/${model.id}:high`, - smol: "pi/default:low", + smol: "@default:low", }); const registry = { getAvailable: () => [model], diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index e2706d95e..278ef70b0 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -213,7 +213,7 @@ describe("createAgentSession deferred model pattern resolution", () => { settings.setModelRole("smol", "runtime-provider/runtime-model"); const { session, modelFallbackMessage } = await createAgentSession({ - ...(await buildSessionOptions("pi/smol")), + ...(await buildSessionOptions("@smol")), settings, }); @@ -265,7 +265,7 @@ describe("createAgentSession deferred model pattern resolution", () => { test("does not apply default role thinking override when modelPattern is explicit", async () => { const settings = Settings.isolated({ defaultThinkingLevel: "off" }); settings.setModelRole("smol", "runtime-provider/runtime-reasoning-model"); - settings.setModelRole("default", "pi/smol:high"); + settings.setModelRole("default", "@smol:high"); const { session } = await createAgentSession({ ...(await buildSessionOptions("runtime-provider/runtime-reasoning-model")), diff --git a/packages/coding-agent/test/status-line-model.test.ts b/packages/coding-agent/test/status-line-model.test.ts index 0db508ec1..d11ec147d 100644 --- a/packages/coding-agent/test/status-line-model.test.ts +++ b/packages/coding-agent/test/status-line-model.test.ts @@ -26,6 +26,7 @@ function createModelContext(advisorActive: boolean): SegmentContext { options: {}, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-overflow.test.ts b/packages/coding-agent/test/status-line-overflow.test.ts index 67a739972..3f5f204c7 100644 --- a/packages/coding-agent/test/status-line-overflow.test.ts +++ b/packages/coding-agent/test/status-line-overflow.test.ts @@ -45,6 +45,7 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null }, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-path.test.ts b/packages/coding-agent/test/status-line-path.test.ts index c32526a59..a238b606e 100644 --- a/packages/coding-agent/test/status-line-path.test.ts +++ b/packages/coding-agent/test/status-line-path.test.ts @@ -31,6 +31,7 @@ function createPathContext(): SegmentContext { }, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-time-spent.test.ts b/packages/coding-agent/test/status-line-time-spent.test.ts index 35620a132..e5dd3df92 100644 --- a/packages/coding-agent/test/status-line-time-spent.test.ts +++ b/packages/coding-agent/test/status-line-time-spent.test.ts @@ -41,6 +41,7 @@ function createCtx(activeMs: number): SegmentContext { options: {}, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/task/executor-pass-through.test.ts b/packages/coding-agent/test/task/executor-pass-through.test.ts index 642d799ca..7a79547a1 100644 --- a/packages/coding-agent/test/task/executor-pass-through.test.ts +++ b/packages/coding-agent/test/task/executor-pass-through.test.ts @@ -175,7 +175,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const result = await runSubprocess({ ...baseOptions, - agent: { ...baseAgent, model: ["pi/task"] }, + agent: { ...baseAgent, model: ["@task"] }, id: "subagent-thinking-precedence", settings, modelRegistry: createModelRegistry(model), @@ -199,7 +199,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const result = await runSubprocess({ ...baseOptions, - agent: { ...baseAgent, model: ["pi/task"] }, + agent: { ...baseAgent, model: ["@task"] }, id: "subagent-thinking-default", settings, modelRegistry: createModelRegistry(model), diff --git a/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts b/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts index 63f8daf39..c193d2914 100644 --- a/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts +++ b/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/sdk"; import { BrowserTool } from "@oh-my-pi/pi-coding-agent/tools/browser"; +import { ensureChromiumExecutable } from "@oh-my-pi/pi-coding-agent/tools/browser/launch"; function makeSession(): ToolSession { return { @@ -13,7 +14,27 @@ function makeSession(): ToolSession { }; } -describe("browser tab evaluation", () => { +/** + * Whether the Chromium puppeteer resolves can actually execute on this host. + * CI runners without Chrome's system libraries (libnspr4 & co.) hold the + * downloaded binary but cannot exec it — probe with --version and skip + * instead of failing. + */ +async function chromiumCanLaunch(): Promise { + try { + const executable = await ensureChromiumExecutable(); + if (!executable) return false; + const probe = Bun.spawnSync([executable, "--version"], { stdout: "ignore", stderr: "ignore" }); + return probe.exitCode === 0; + } catch { + return false; + } +} + +const CHROMIUM_AVAILABLE = await chromiumCanLaunch(); + +describe.skipIf(!CHROMIUM_AVAILABLE)("browser tab evaluation", () => { + // Launches real headless Chromium; CI cold start easily exceeds bun's 5s default. it("runs tab.evaluate in the page's main JavaScript world", async () => { const tool = new BrowserTool(makeSession()); const name = `main-world-${process.pid}`; @@ -34,5 +55,5 @@ describe("browser tab evaluation", () => { } finally { await tool.execute("close", { action: "close", name, kill: true }); } - }); + }, 30_000); }); diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index 7b84723cf..7cee7fcb6 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -57,34 +57,54 @@ describe("editToolRenderer", () => { expect(rendered).toContain("packages/coding-agent/src/edit/renderer.ts"); }); - it("lifts the streaming diff tail window when expanded", async () => { + it("windows the expanded streaming diff to the viewport tail", async () => { const uiTheme = await getUiTheme(); - const diff = Array.from({ length: 20 }, (_, index) => - index === 0 ? "-head-line-1" : `+tail-line-${index + 1}`, - ).join("\n"); - const renderPreview = (expanded: boolean): string => - Bun.stripANSI( - editToolRenderer - .renderCall( - { file_path: "/tmp/preview.ts", previewDiff: diff }, - { expanded, isPartial: true, spinnerFrame: 0, renderContext: { editMode: "replace" } }, - uiTheme, - ) - .render(200) - .join("\n"), - ); + // Pin a tall viewport so previewWindowRows() (rows - reserve) lands at 30: + // collapsed stays at the 12-row fixed tail, expanded widens to 30. + const originalRowsDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "rows"); + Object.defineProperty(process.stdout, "rows", { value: 50, configurable: true }); + try { + const makeDiff = (length: number): string => + Array.from({ length }, (_, index) => (index === 0 ? "-head-line-1" : `+tail-line-${index + 1}`)).join("\n"); + const renderPreview = (diff: string, expanded: boolean): string => + Bun.stripANSI( + editToolRenderer + .renderCall( + { file_path: "/tmp/preview.ts", previewDiff: diff }, + { expanded, isPartial: true, spinnerFrame: 0, renderContext: { editMode: "replace" } }, + uiTheme, + ) + .render(200) + .join("\n"), + ); - const collapsed = renderPreview(false); - expect(collapsed).toContain("tail-line-20"); - expect(collapsed).not.toContain("head-line-1"); - expect(collapsed).toContain("more lines above"); - expect(collapsed).toContain("(preview)"); + const collapsed = renderPreview(makeDiff(20), false); + expect(collapsed).toContain("tail-line-20"); + expect(collapsed).not.toContain("head-line-1"); + expect(collapsed).toContain("more lines above"); + expect(collapsed).toContain("(preview)"); - const expanded = renderPreview(true); - expect(expanded).toContain("head-line-1"); - expect(expanded).toContain("tail-line-20"); - expect(expanded).not.toContain("more lines above"); - expect(expanded).not.toContain("(preview)"); + // Within the viewport window, expanded shows the whole diff. + const expanded = renderPreview(makeDiff(20), true); + expect(expanded).toContain("head-line-1"); + expect(expanded).toContain("tail-line-20"); + expect(expanded).not.toContain("more lines above"); + expect(expanded).not.toContain("(preview)"); + + // Beyond it, expanded stays a viewport-sized tail window: an unbounded + // live preview scrolls above the native-scrollback commit boundary and + // freezes a stale snapshot that duplicates the block at finalize. + const expandedTall = renderPreview(makeDiff(40), true); + expect(expandedTall).toContain("tail-line-40"); + expect(expandedTall).not.toContain("head-line-1"); + expect(expandedTall).toContain("more lines above"); + } finally { + if (originalRowsDescriptor) { + Object.defineProperty(process.stdout, "rows", originalRowsDescriptor); + } else { + Reflect.deleteProperty(process.stdout, "rows"); + } + } }); it("uses hashline input headers for streaming call path without apply_patch errors", async () => { diff --git a/packages/coding-agent/test/tools/inspect-image.test.ts b/packages/coding-agent/test/tools/inspect-image.test.ts index 91dfcc661..bfdced519 100644 --- a/packages/coding-agent/test/tools/inspect-image.test.ts +++ b/packages/coding-agent/test/tools/inspect-image.test.ts @@ -352,7 +352,7 @@ describe("InspectImageTool", () => { expect(stub.calls).toHaveLength(0); }); - it("falls back to pi/default when vision role is unset", async () => { + it("falls back to @default when vision role is unset", async () => { const imagePath = path.join(testDir, "screen.png"); fs.writeFileSync(imagePath, Buffer.from(TINY_PNG_BASE64, "base64")); diff --git a/packages/coding-agent/test/tools/launch-renderer.test.ts b/packages/coding-agent/test/tools/launch-renderer.test.ts index e9c2723b3..fea9522ce 100644 --- a/packages/coding-agent/test/tools/launch-renderer.test.ts +++ b/packages/coding-agent/test/tools/launch-renderer.test.ts @@ -6,6 +6,7 @@ */ import { describe, expect, it } from "bun:test"; import type { DaemonSnapshot } from "@oh-my-pi/pi-coding-agent/launch/protocol"; +import { renderTerminalOutput } from "@oh-my-pi/pi-coding-agent/launch/terminal-output"; import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { type LaunchToolDetails, launchToolRenderer } from "@oh-my-pi/pi-coding-agent/tools/launch"; import { toolRenderers } from "@oh-my-pi/pi-coding-agent/tools/renderers"; @@ -78,9 +79,46 @@ describe("launchToolRenderer", () => { ); expect(rendered[0]).toContain("Launch logs"); expect(rendered[0]).toContain("cursor 2210"); - expect(rendered).toContain("line one"); - expect(rendered).toContain("line two"); + expect(rendered.some(line => line.includes("line one"))).toBe(true); + expect(rendered.some(line => line.includes("line two"))).toBe(true); expect(rendered.some(line => line.includes("[web: running"))).toBe(false); + expect(rendered[0]).toContain("╭"); + expect(rendered.some(line => line.includes("Output"))).toBe(true); + expect(rendered.at(-1)).toContain("╰"); + }); + + it("replays terminal screen rows so cursor rewrites retain their final color and weight", async () => { + const terminalRows = await renderTerminalOutput( + "\x1b[1;31mold\x1b[0m\r\x1b[2K\x1b[12G\x1b[1;32mready\x1b[0m\x1b[K", + { + head: false, + maxRows: 10, + }, + ); + if (terminalRows === undefined) throw new Error("terminal replay failed"); + expect(Bun.stripANSI(terminalRows[0] ?? "")).toBe(" ready"); + + const uiTheme = await theme(); + const component = launchToolRenderer.renderResult( + { + content: [{ type: "text", text: "ready\n[web: running; cursor=2210]" }], + details: { + op: "logs", + cursor: 2210, + timedOut: false, + state: "running", + terminalRows, + } satisfies LaunchToolDetails, + }, + { expanded: false, isPartial: false }, + uiTheme, + { op: "logs", name: "web" }, + ); + const raw = component.render(200).join("\n"); + const plain = Bun.stripANSI(raw); + expect(plain).toContain("ready"); + expect(plain).not.toContain("old"); + expect(raw).toContain("\x1b[1;38;5;2mready"); }); it("caps a collapsed list to the preview item limit with a more-items row", async () => { diff --git a/packages/coding-agent/test/tools/launch.test.ts b/packages/coding-agent/test/tools/launch.test.ts index eb5c95f3e..df0bb0bda 100644 --- a/packages/coding-agent/test/tools/launch.test.ts +++ b/packages/coding-agent/test/tools/launch.test.ts @@ -59,7 +59,9 @@ describe("daemon broker", () => { `process.stdin.setRawMode?.(true); process.stdin.setEncoding("utf8"); process.stdin.resume(); -process.stdout.write("READY\\n"); +process.stdout.write("\\x1b[2J\\x1b[H"); +for (let index = 0; index < 25; index++) process.stdout.write("BOOT:" + index + "\\n"); +process.stdout.write("\\x1b[1;32mREADY\\x1b[0m\\n"); process.stdin.on("data", chunk => process.stdout.write("INPUT:" + JSON.stringify(chunk) + "\\n")); setInterval(() => {}, 1000); `, @@ -114,7 +116,12 @@ setInterval(() => {}, 1000); expect(logs.op).toBe("logs"); if (logs.op !== "logs") throw new Error("unexpected logs result"); expect(logs.text).toContain("READY"); + expect(logs.text).not.toContain("\x1b"); + expect(logs.text).not.toContain("BOOT:0"); expect(logs.text).toContain('INPUT:"run\\r"'); + expect(logs.terminalText).toContain("\x1b[2J\x1b[H"); + expect(logs.terminalText).toContain("\x1b[1;32mREADY\x1b[0m"); + expect(logs.terminalText).toContain("BOOT:0"); const stopped = await first.request({ op: "stop", name: "debugger", timeoutMs: 2_000 }); expect(stopped.op).toBe("stop"); diff --git a/packages/harbor-manager/scripts/dev.ts b/packages/harbor-manager/scripts/dev.ts deleted file mode 100755 index 5cd7c102b..000000000 --- a/packages/harbor-manager/scripts/dev.ts +++ /dev/null @@ -1,39 +0,0 @@ -#!/usr/bin/env bun -/** - * Dev harness: runs the Bun API server (auto-restarting on server edits via - * `--watch`) and a Vite dev server (React Fast Refresh for the dashboard) - * together, tearing both down on one Ctrl-C. Vite proxies `/api` to the API - * server; the shared port travels through `HARBOR_API_PORT`. - * - * Extra args pass through to the API server: - * bun run dev -- --port 4700 --jobs-dir ../../runs/harbor - * - * Vite runs under Node (its bin shebang), the API under Bun; only the frontend - * hot-reloads in place, while server-side changes trigger a fast `--watch` restart. - */ -const args = Bun.argv.slice(2); -const portIndex = args.indexOf("--port"); -const apiPort = portIndex >= 0 ? (args[portIndex + 1] ?? "4700") : "4700"; -process.env.HARBOR_API_PORT = apiPort; - -const io = { stdout: "inherit", stderr: "inherit", stdin: "inherit", env: { ...process.env } } as const; -const api = Bun.spawn(["bun", "--watch", "src/server.ts", ...args], io); -const web = Bun.spawn(["vite"], io); - -let stopping = false; -const stop = (): void => { - if (stopping) return; - stopping = true; - try { - api.kill(); - } catch {} - try { - web.kill(); - } catch {} -}; -process.on("SIGINT", stop); -process.on("SIGTERM", stop); - -await Promise.race([api.exited, web.exited]); -stop(); -process.exit(0); diff --git a/packages/harbor-manager/src/runner.test.ts b/packages/harbor-manager/src/runner.test.ts deleted file mode 100644 index 1b5e53431..000000000 --- a/packages/harbor-manager/src/runner.test.ts +++ /dev/null @@ -1,129 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { buildHarborEnv, collectForwardEnv, parseArgs } from "./runner"; - -describe("generic agent-arg / env passthrough", () => { - it("forwards repeated --agent-arg as a JSON array the in-container agent can parse", () => { - const cfg = parseArgs([ - "--model", - "anthropic/claude-opus-4-8", - "--agent-arg", - "--downshift", - "--agent-arg", - "--downshift-into", - "--agent-arg", - "google/gemini-3.5-flash", - ]); - expect(cfg.agentArgs).toEqual(["--downshift", "--downshift-into", "google/gemini-3.5-flash"]); - - const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); - expect(JSON.parse(env.OMP_BENCH_AGENT_ARGS ?? "[]")).toEqual(cfg.agentArgs); - }); - - it("omits OMP_BENCH_AGENT_ARGS when no --agent-arg was passed", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); - const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); - expect(env.OMP_BENCH_AGENT_ARGS).toBeUndefined(); - }); - - it("routes an explicit --providers entry alongside the model's own provider", () => { - // The runner has no built-in concept of a "second model"; gateway routing - // for any extra model introduced via --agent-arg must be declared - // explicitly via --providers. - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--providers", "google"]); - const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); - expect(new Set(env.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["anthropic", "google"])); - }); - - it("collects explicit --env pairs, with an explicit value winning over a bare host-forwarded key", () => { - const cfg = parseArgs([ - "--model", - "anthropic/claude-opus-4-8", - "--env", - "SOME_FLAG=1", - "--env", - "OTHER=two words", - ]); - const forwarded = collectForwardEnv(cfg); - expect(forwarded.SOME_FLAG).toBe("1"); - expect(forwarded.OTHER).toBe("two words"); - }); -}); - -describe("install modes", () => { - it("defaults to source mode and publishes the mount contract to the agent", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); - expect(cfg.install).toBe("source"); - const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test", { - arch: "arm64", - depsDir: "/tmp/deps", - nodeModules: ["node_modules"], - }); - expect(env.OMP_BENCH_INSTALL).toBe("source"); - expect(env.OMP_BENCH_SOURCE_DIR).toBe("/opt/omp/src"); - expect(env.OMP_BENCH_SOURCE_BUN).toBe("/opt/omp/bin/bun"); - expect(env.OMP_BENCH_SOURCE_ARCH).toBe("arm64"); - }); - - it("omits source mount env when no mount was prepared (binary/local runs)", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--install", "local"]); - const env = buildHarborEnv(cfg, "/tmp/models.yml", "/tmp/omp.tgz", "test"); - expect(env.OMP_BENCH_INSTALL).toBe("local"); - expect(env.OMP_BENCH_SOURCE_DIR).toBeUndefined(); - expect(env.OMP_BENCH_SOURCE_ARCH).toBeUndefined(); - }); - - it("--tarball implies a local (tarball) install", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--tarball", "/tmp/omp.tgz"]); - expect(cfg.install).toBe("local"); - expect(cfg.build).toBe(false); - }); -}); - -describe("parseArgs validation", () => { - it("rejects an unknown flag", () => { - expect(() => parseArgs(["--model", "anthropic/claude-opus-4-8", "--not-a-real-flag"])).toThrow(/unknown flag/); - }); - - it("defaults to a generic, dataset-agnostic jobs directory", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); - expect(cfg.jobsDir.endsWith("/runs/harbor")).toBe(true); - }); -}); - -describe("environment backends", () => { - it("defaults to docker with the host.docker.internal gateway", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); - expect(cfg.envType).toBe("docker"); - expect(cfg.gatewayUrl).toBe("http://host.docker.internal:4000"); - }); - - it("apple-container swaps the default gateway host to the vmnet bridge address", () => { - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "apple-container"]); - expect(cfg.envType).toBe("apple-container"); - expect(cfg.gatewayUrl).toBe("http://192.168.64.1:4000"); - }); - - it("an explicit --gateway-url wins over the apple-container default, regardless of flag order", () => { - const cfg = parseArgs([ - "--model", - "anthropic/claude-opus-4-8", - "--gateway-url", - "http://10.0.0.5:9999", - "--environment", - "apple-container", - ]); - expect(cfg.gatewayUrl).toBe("http://10.0.0.5:9999"); - }); - - it("rejects --host-network with apple-container (compose overlay is docker-only)", () => { - expect(() => - parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "apple-container", "--host-network"]), - ).toThrow(/docker-only/); - }); - - it("rejects an invalid --environment value", () => { - expect(() => parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "podman"])).toThrow( - /--environment must be/, - ); - }); -}); diff --git a/packages/harbor-manager/vite.config.ts b/packages/harbor-manager/vite.config.ts deleted file mode 100644 index 063604c6e..000000000 --- a/packages/harbor-manager/vite.config.ts +++ /dev/null @@ -1,24 +0,0 @@ -import react from "@vitejs/plugin-react"; -import { defineConfig } from "vite"; - -/** - * Dev-only Vite config for the harbor-manager dashboard. - * - * Serves `src/web` with React Fast Refresh and proxies the API — including the - * SSE run stream at `/api/events` — to the Bun server that `scripts/dev.ts` - * starts alongside it (`HARBOR_API_PORT`, default 4700). Production is served by - * the Bun server itself (`src/server.ts` bundles `app.tsx` on demand); Vite is - * not part of the production path. - */ -const apiTarget = `http://localhost:${process.env.HARBOR_API_PORT ?? "4700"}`; - -export default defineConfig({ - root: "src/web", - server: { - port: Number(process.env.HARBOR_WEB_PORT ?? "5173"), - proxy: { - "/api": { target: apiTarget, changeOrigin: true }, - }, - }, - plugins: [react()], -}); diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index bd0e21a57..151d36fad 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -2,10 +2,13 @@ ## [Unreleased] +## [16.5.0] - 2026-07-13 + ### Fixed -- Rejected ambiguous swaps that risk silent deletion of range boundaries -- Prevented ambiguous auto-repairing of structural closing lines when payload placement is unclear +- Fixed a critical issue where ambiguous swaps could silently delete range boundaries. +- Prevented incorrect auto-repairing of structural closing lines when payload placement is ambiguous. +- Fixed a bug in stale-hash recovery that could incorrectly relocate edits onto duplicated context after the original target changed. ## [16.3.3] - 2026-07-02 diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 208dcaaac..f348da275 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.4.8", + "version": "16.5.0", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/src/messages.ts b/packages/hashline/src/messages.ts index 5c93c70e7..458585143 100644 --- a/packages/hashline/src/messages.ts +++ b/packages/hashline/src/messages.ts @@ -208,15 +208,6 @@ export const RECOVERY_EXTERNAL_WARNING = export const RECOVERY_SESSION_CHAIN_WARNING = "Recovered from a stale file hash using an earlier in-session snapshot (a prior edit in this session advanced the hash)."; -/** - * `Recovery`: session-chain replay fast-path. Less certain than - * {@link RECOVERY_SESSION_CHAIN_WARNING} — the 3-way merge refused, the - * anchor-content gate passed, but a coincidental insert+delete earlier in - * the chain could still misplace an anchor — hence the verify hedge. - */ -export const RECOVERY_SESSION_REPLAY_WARNING = - "Recovered by replaying your edits onto the current file content (a prior in-session edit changed the lines you re-targeted with a stale hash). Verify the diff matches your intent."; - /** `Recovery`: stale anchors were relocated to unchanged live lines after drift. */ export const RECOVERY_LINE_REMAP_WARNING = "Recovered by remapping stale line anchors to unchanged current lines (file changed since the tagged read). Verify the diff matches your intent."; diff --git a/packages/hashline/src/patcher.ts b/packages/hashline/src/patcher.ts index d4d98488d..b49156135 100644 --- a/packages/hashline/src/patcher.ts +++ b/packages/hashline/src/patcher.ts @@ -586,7 +586,7 @@ export class Patcher { const expected = exists ? section.fileHash : undefined; // The 4-hex tag is content-derived: when the live text hashes to it, // trust the match and apply directly. `storedSnapshotForTag` feeds the - // drift paths below (block resolution, 3-way recovery); on a 16-bit + // drift paths below (block resolution, anchor remapping); on a 16-bit // tag collision it resolves to the most-recently recorded text. const storedSnapshotForTag = expected === undefined ? null : this.snapshots.byHash(canonicalPath, expected); const liveMatches = expected !== undefined && computeFileHash(normalized) === expected; @@ -598,7 +598,7 @@ export class Patcher { // - live content matches the tag (or there is no tag) → resolve against // the live, normalized content; // - the file drifted → resolve against the tagged snapshot's text so the - // resulting ranges flow through the 3-way-merge recovery below. + // resulting ranges can be mapped to unchanged live lines below. // When a block edit needs the tagged snapshot but it is unavailable, the // range cannot be placed safely — reject with a MismatchError (re-read). const blockResolutions: BlockResolution[] = []; @@ -640,8 +640,8 @@ export class Patcher { const result = applyEdits(normalized, resolved); return withResolveWarnings({ ...result, warnings: [HEADTAIL_DRIFT_WARNING, ...(result.warnings ?? [])] }); } - // File drifted: try to replay the edit against the version the tag - // names and 3-way-merge it onto the live content. + // File drifted: map every anchor from the tagged snapshot to unchanged + // live lines. Recovery refuses changed or ambiguous targets. const recovered = this.recovery.tryRecover({ path: canonicalPath, currentText: normalized, diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts index b3e1b9ff4..eb48cee9e 100644 --- a/packages/hashline/src/recovery.ts +++ b/packages/hashline/src/recovery.ts @@ -1,29 +1,17 @@ /** - * Recover from a stale section snapshot tag by replaying the would-be edit - * against a cached pre-edit snapshot of the file and 3-way-merging the - * result onto the current on-disk content. + * Recovers stale section tags by proving that every anchored line still maps + * to one unchanged, contiguous region in the current file, then replaying the + * edit against that live content. * - * The patcher consults this when a section tag resolves to a snapshot that no - * longer matches the live file content. The recovery class is stateless apart - * from the {@link SnapshotStore} it queries; the snapshot store is the seam - * lets you plug in your own caching strategy. + * Recovery fails closed when the target changed or became ambiguous. The + * patcher then returns a mismatch with fresh context instead of guessing. */ import * as Diff from "diff"; import { applyEdits } from "./apply"; -import { - RECOVERY_EXTERNAL_WARNING, - RECOVERY_LINE_REMAP_WARNING, - RECOVERY_SESSION_CHAIN_WARNING, - RECOVERY_SESSION_REPLAY_WARNING, -} from "./messages"; -import type { Snapshot, SnapshotStore } from "./snapshots"; +import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages"; +import type { SnapshotStore } from "./snapshots"; import type { Anchor, ApplyResult, Edit } from "./types"; -// Section tags are line-precise; never let Diff.applyPatch slide a hunk -// onto a duplicate closer 100+ lines away. If snapshot replay does not -// align exactly, refuse and let the caller re-read. -const RECOVERY_FUZZ_FACTOR = 0; - export interface RecoveryArgs { path: string; currentText: string; @@ -40,31 +28,6 @@ export interface RecoveryResult { warnings: string[]; } -function applyEditsToSnapshot( - previousText: string, - currentText: string, - edits: readonly Edit[], - recoveryWarning: string, -): RecoveryResult | null { - let applied: ApplyResult; - try { - applied = applyEdits(previousText, [...edits]); - } catch { - return null; - } - if (applied.text === previousText) return null; - - const patch = Diff.structuredPatch("file", "file", previousText, applied.text, "", "", { context: 3 }); - const merged = Diff.applyPatch(currentText, patch, { fuzzFactor: RECOVERY_FUZZ_FACTOR }); - if (typeof merged !== "string" || merged === currentText) return null; - - const firstChangedLine = findFirstChangedLine(currentText, merged) ?? applied.firstChangedLine; - const hasNetChange = firstChangedLine !== undefined; - const warnings = hasNetChange ? [recoveryWarning, ...(applied.warnings ?? [])] : [...(applied.warnings ?? [])]; - - return { text: merged, firstChangedLine, warnings }; -} - function collectAnchorLines(edits: readonly Edit[]): number[] { const lines: number[] = []; for (const edit of edits) { @@ -81,27 +44,6 @@ function getEditAnchors(edit: Edit): Anchor[] { return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : []; } -/** - * Returns true when every anchor line in `edits` has identical content in - * `previousText` and `currentText`. The session-chain replay fast-path - * requires this: if the prior in-session edit rewrote the line the model is - * now re-targeting with a stale hash, replaying onto current would silently - * overwrite the new content with whatever the model authored against the - * old content — a corruption window, not a recovery. - */ -function verifyAnchorContent(previousText: string, currentText: string, edits: readonly Edit[]): boolean { - const lines = collectAnchorLines(edits); - if (lines.length === 0) return true; - const prev = previousText.split("\n"); - const curr = currentText.split("\n"); - for (const line of lines) { - const idx = line - 1; - if (idx < 0 || idx >= prev.length || idx >= curr.length) return false; - if (prev[idx] !== curr[idx]) return false; - } - return true; -} - function buildLineMap(previousText: string, currentText: string): Map { const previousLines = previousText.split("\n"); const currentLines = currentText.split("\n"); @@ -198,7 +140,7 @@ function validateUniqueAnchorContext( ): boolean { const offset = mapped - line; const { before, after } = neighbors; - if (after !== undefined) return lineMap.get(after) === after + offset; + if (after !== undefined && lineMap.get(after) === after + offset) return true; return before !== undefined && lineMap.get(before) === before + offset; } @@ -237,7 +179,12 @@ function validateRemappedAnchorContext( return true; } -function remapEditsToCurrent(previousText: string, currentText: string, edits: readonly Edit[]): Edit[] | null { +interface RemappedEdits { + edits: Edit[]; + offset: number; +} + +function remapEditsToCurrent(previousText: string, currentText: string, edits: readonly Edit[]): RemappedEdits | null { const lineMap = buildLineMap(previousText, currentText); if (!validateRemappedAnchorContext(previousText, currentText, lineMap, edits)) return null; const offsets: number[] = []; @@ -289,21 +236,21 @@ function remapEditsToCurrent(previousText: string, currentText: string, edits: r if (offsets.length === 0) return null; const firstOffset = offsets[0]; - if (firstOffset === 0) return null; if (!offsets.every(offset => offset === firstOffset)) return null; - return remapped; + return { edits: remapped, offset: firstOffset }; } function replayRemappedAnchorsOnCurrent( previousText: string, currentText: string, edits: readonly Edit[], + recoveryWarning: string, ): RecoveryResult | null { const remapped = remapEditsToCurrent(previousText, currentText, edits); if (remapped === null) return null; let applied: ApplyResult; try { - applied = applyEdits(currentText, remapped); + applied = applyEdits(currentText, remapped.edits); } catch { return null; } @@ -311,78 +258,18 @@ function replayRemappedAnchorsOnCurrent( return { text: applied.text, firstChangedLine: applied.firstChangedLine, - warnings: [RECOVERY_LINE_REMAP_WARNING, ...(applied.warnings ?? [])], + warnings: [remapped.offset === 0 ? recoveryWarning : RECOVERY_LINE_REMAP_WARNING, ...(applied.warnings ?? [])], }; } - -function replaySessionChainOnCurrent( - previousText: string, - currentText: string, - edits: readonly Edit[], -): RecoveryResult | null { - // Two guards narrow the corruption window. Neither alone is sufficient, - // and even together they don't fully prove correctness — replay is the - // less-certain recovery mode and emits RECOVERY_SESSION_REPLAY_WARNING - // so the caller can verify the diff. - // - Equal line counts: every line number in `edits` still resolves to - // SOME logical row (no net shift across the prior chain). A - // coincidental insert+delete pair can still leave indices pointing - // at different logical rows than the model anchored against. - // - Anchor-content alignment: the row at each anchor's line index has - // identical content in previous and current. Catches the common - // case of a prior edit rewriting the targeted line; can still be - // coincidentally satisfied by a duplicated row at the shifted - // index. - if (previousText.split("\n").length !== currentText.split("\n").length) return null; - if (!verifyAnchorContent(previousText, currentText, edits)) return null; - let applied: ApplyResult; - try { - applied = applyEdits(currentText, [...edits]); - } catch { - return null; - } - if (applied.text === currentText) return null; - return { - text: applied.text, - firstChangedLine: applied.firstChangedLine, - warnings: [RECOVERY_SESSION_REPLAY_WARNING, ...(applied.warnings ?? [])], - }; -} - -/** First 1-indexed line at which `a` and `b` diverge, or `undefined` if equal. */ -function findFirstChangedLine(a: string, b: string): number | undefined { - if (a === b) return undefined; - const aLines = a.split("\n"); - const bLines = b.split("\n"); - const max = Math.max(aLines.length, bLines.length); - for (let i = 0; i < max; i++) { - if (aLines[i] !== bLines[i]) return i + 1; - } - return undefined; -} - -function isHeadSnapshot(head: Snapshot | null, snapshot: Snapshot): boolean { - return head === snapshot; -} - /** * Stateless recovery driver over a {@link SnapshotStore}. Construct once and - * call {@link Recovery.tryRecover} per stale-tag incident. The default - * implementation tries three strategies in order: + * call {@link Recovery.tryRecover} per stale-tag incident. * - * 1. Apply the edits on the full-file version the tag names, then 3-way-merge - * the resulting patch onto the live content (handles external writes). - * 2. Remap every stale anchor through the unchanged-line diff from the tagged - * snapshot to the live text, then replay on live content. This handles a - * prior insertion/deletion before the target while refusing changed anchors - * and mixed offsets across the same edit range. - * 3. (Session chain) If that version wasn't the head, replay the edits onto - * the live content directly when line counts match AND every edit's anchor - * line content is unchanged between version and current — a prior in-session - * edit advanced the tag and the model's anchors still name the same logical - * rows. Emits a dedicated {@link RECOVERY_SESSION_REPLAY_WARNING} because - * even with both guards a coincidental insert+delete pair on duplicate rows - * can still land the edit on the wrong row; see {@link replaySessionChainOnCurrent}. + * Recovery maps every stale anchor through unchanged lines from the tagged + * snapshot to the live text, validates surrounding context, and replays the + * edit directly on live content. All anchors must move by one consistent + * offset. A changed, deleted, split, or ambiguous target is rejected so the + * caller can surface a {@link MismatchError} with current context. */ export class Recovery { constructor(readonly store: SnapshotStore) {} @@ -392,26 +279,12 @@ export class Recovery { */ tryRecover(args: RecoveryArgs): RecoveryResult | null { const { path, currentText, fileHash, edits } = args; - // When two retained texts collide on the 16-bit tag, resolve to the - // most-recently recorded one; a wrong pick can only land if one of the - // merge/remap/session-chain strategies below applies it cleanly. + // When retained texts collide on the 16-bit tag, use the latest one. + // Recovery still requires its anchors and context to map unambiguously. const snapshot = this.store.byHash(path, fileHash); if (!snapshot) return null; - const isHead = isHeadSnapshot(this.store.head(path), snapshot); - const recoveryWarning = isHead ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; - const merged = applyEditsToSnapshot(snapshot.text, currentText, edits, recoveryWarning); - if (merged !== null) return merged; - // Line-shift fallback: the 3-way merge refused, but unchanged anchor - // lines may have moved because a prior edit inserted or deleted rows - // before them. Remap only when every anchor resolves through the diff - // with one consistent offset; otherwise the edit range was touched. - const remapped = replayRemappedAnchorsOnCurrent(snapshot.text, currentText, edits); - if (remapped !== null) return remapped; - // Session-chain fallback: replay onto current is gated by line-count - // equality AND anchor-content alignment — see - // `replaySessionChainOnCurrent` for why both guards together still - // don't fully prove correctness. - if (!isHead) return replaySessionChainOnCurrent(snapshot.text, currentText, edits); - return null; + const recoveryWarning = + this.store.head(path) === snapshot ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; + return replayRemappedAnchorsOnCurrent(snapshot.text, currentText, edits, recoveryWarning); } } diff --git a/packages/hashline/src/snapshots.ts b/packages/hashline/src/snapshots.ts index 97189ec3b..874433b2b 100644 --- a/packages/hashline/src/snapshots.ts +++ b/packages/hashline/src/snapshots.ts @@ -11,8 +11,7 @@ * {@link SnapshotStore.record} with the full normalized text they observed. * The store hashes it, dedups against the per-path history, and returns the * tag. Consumers (recovery, the patcher) resolve a stale tag back to the - * recorded full text via {@link SnapshotStore.byHash} and 3-way-merge the - * would-be edit onto the live content. + * recorded full text and map its unchanged edit anchors onto live content. * * The abstract base class lets callers plug in whatever storage they like * (LRU, persistent SQLite, etc.). {@link InMemorySnapshotStore} ships as a @@ -199,8 +198,8 @@ export class InMemorySnapshotStore extends SnapshotStore { // texts that happen to share the 4-hex tag are DIFFERENT snapshots — fusing // them under one entry would corrupt seenLines (attaching lines from // text B onto the stored text A) and let the patcher misresolve which - // snapshot the section tag names when it does 3-way merge or seen-line - // validation. See issue #4075. + // snapshot the section tag names during recovery or seen-line validation. + // See issue #4075. const existing = history.find(version => version.hash === hash && version.text === fullText); if (existing) { // Same content state observed again: refresh recency and promote to diff --git a/packages/hashline/test/block.test.ts b/packages/hashline/test/block.test.ts index d6eb03b32..2526198b7 100644 --- a/packages/hashline/test/block.test.ts +++ b/packages/hashline/test/block.test.ts @@ -229,7 +229,7 @@ describe("Patcher with a block resolver", () => { const patcher = new Patcher({ fs, snapshots, blockResolver: stubResolver }); // `block 2` resolves against the SNAPSHOT → span [2,3] → replace - // "line1","line2"; recovery 3-way-merges the change onto the live file. + // "line1","line2"; recovery maps that unchanged span onto the live file. const result = await patcher.apply(Patch.parse(`[${PATH}#${tag}]\nSWAP.BLK 2:\n+NEW`)); expect(result.sections[0]?.op).toBe("update"); diff --git a/packages/hashline/test/boundary-repair.test.ts b/packages/hashline/test/boundary-repair.test.ts index cdf90aa35..6c53e9a06 100644 --- a/packages/hashline/test/boundary-repair.test.ts +++ b/packages/hashline/test/boundary-repair.test.ts @@ -607,8 +607,8 @@ describe("boundary-balance repair through stale-snapshot recovery", () => { // Recovery composes `applyEdits` to compute the intended change, so the // boundary repair runs there too. The snapshot (what the model read) // carries the structure; the live file has drifted far from the edit - // region, so the stale-hash 3-way merge succeeds and the repaired - // (de-duplicated) hunk lands without doubling the closer. + // region, so anchor recovery succeeds and the repaired (de-duplicated) + // hunk lands without doubling the closer. it("de-duplicates a closer while recovering from a drifted file", () => { const snapshotLines = [ 'import { x } from "y";', @@ -628,7 +628,7 @@ describe("boundary-balance repair through stale-snapshot recovery", () => { ]; const snapshotText = `${snapshotLines.join("\n")}\n`; // Live file drifted only at the tail (line 13) — far outside the edit - // region (lines 4-6), so the 3-way merge applies cleanly. + // region (lines 4-6), so unchanged-anchor recovery succeeds. const currentText = snapshotText.replace("const tail = 0;", "const tail = 99;"); const store = new InMemorySnapshotStore(); diff --git a/packages/hashline/test/recovery-session-chain.test.ts b/packages/hashline/test/recovery-session-chain.test.ts index 08db6634e..1fbbb85e9 100644 --- a/packages/hashline/test/recovery-session-chain.test.ts +++ b/packages/hashline/test/recovery-session-chain.test.ts @@ -15,7 +15,7 @@ import { InMemorySnapshotStore, parsePatch, RECOVERY_LINE_REMAP_WARNING, - RECOVERY_SESSION_REPLAY_WARNING, + RECOVERY_SESSION_CHAIN_WARNING, Recovery, } from "@oh-my-pi/hashline"; @@ -58,10 +58,9 @@ describe("Recovery — session-chain replay anchor-content gate", () => { it("replays edits onto current when every anchor's line content is unchanged", () => { const { store, v1Text, h0 } = seedTwoSnapshots(); - // Edit anchored at line 3 — unchanged between v0 and v1. The 3-way - // merge fails (patch context includes the rewritten line 5), but the - // replay fallback is safe because the model's anchor still names the - // same logical content. + // Edit anchored at line 3 — unchanged between v0 and v1. Recovery + // proves that the target and its surrounding context still map to the + // same live lines before replaying the edit. const { edits } = parsePatch("SWAP 3.=3:\n|L3-MODEL"); const recovered = new Recovery(store).tryRecover({ @@ -76,11 +75,10 @@ describe("Recovery — session-chain replay anchor-content gate", () => { // Prior in-session change must survive — the model's edit lands on // top of current, not on top of the stale snapshot. expect(recovered?.text).toContain("L5-CHANGED"); - // The replay path is the less-certain recovery mode (a coincidental - // insert+delete pair earlier in the chain could leave indices - // pointing at duplicated rows even with both guards satisfied), so - // the dedicated REPLAY warning surfaces a "verify the diff" hedge. - expect(recovered?.warnings).toContain(RECOVERY_SESSION_REPLAY_WARNING); + // Zero-offset recovery against an earlier retained snapshot reports the + // session-chain banner; unlike the removed direct replay fallback, this + // path has proved the anchors through the unchanged-line map. + expect(recovered?.warnings).toContain(RECOVERY_SESSION_CHAIN_WARNING); }); it("recovers stale anchors shifted by a prior in-session insertion", () => { @@ -141,11 +139,30 @@ describe("Recovery — session-chain replay anchor-content gate", () => { expect(recovered).toBeNull(); }); - it("refuses unique-line remaps when following context no longer matches", () => { + it("refuses to relocate a stale replacement onto duplicated context", () => { + const store = new InMemorySnapshotStore(); + const block = ["head", "TARGET_A", "TARGET_B", "ctx1", "ctx2", "ctx3"]; + const v0Text = lines(...block, "middle", ...block, "tail"); + const hash = store.record(PATH, v0Text); + const currentText = lines("head", "CHANGED_A", "CHANGED_B", "ctx1", "ctx2", "ctx3", "middle", ...block, "tail"); + const { edits } = parsePatch("SWAP 2.=3:\n+MODEL_A\n+MODEL_B"); + + const recovered = new Recovery(store).tryRecover({ + path: PATH, + currentText, + fileHash: hash, + edits, + }); + + expect(recovered).toBeNull(); + expect(currentText).toContain("TARGET_A\nTARGET_B"); + }); + + it("refuses an isolated unique-line remap when neither neighbor follows its offset", () => { const store = new InMemorySnapshotStore(); const v0Text = lines("L1", "L2", "L3", "L4", "T", "L6"); const h0 = store.record(PATH, v0Text); - const v1Text = lines("X", "L1", "L2", "L3", "L4", "T", "T_CHANGED", "L6"); + const v1Text = lines("X", "L1", "L2", "L3", "L4", "BEFORE", "T", "AFTER", "L6"); store.record(PATH, v1Text); const { edits } = parsePatch("SWAP 5.=5:\n+MODEL"); @@ -214,8 +231,8 @@ describe("Recovery — colliding snapshot tags", () => { store.record(PATH, newer); // Live drifted away from both colliders, so recovery cannot shortcut - // via live==snapshot. The tag cannot name a unique base; it resolves - // to the most-recently recorded collider and 3-way merges from there. + // via live==snapshot. The tag cannot name a unique base; recovery uses + // the most-recently retained collider and maps its unchanged anchors. const currentText = `${newer}drifted trailer\n`; const recovered = new Recovery(store).tryRecover({ path: PATH, @@ -228,8 +245,7 @@ describe("Recovery — colliding snapshot tags", () => { }); it("still recovers when exactly one retained text carries the tag", () => { - // Same drift scenario with a single retained text for the tag: the - // plain 3-way merge path. + // Same drift scenario with a single retained text for the tag. const { older } = findCollidingTexts(); const store = new InMemorySnapshotStore(); const tag = store.record(PATH, older); diff --git a/packages/harbor-manager/README.md b/packages/metaharness/README.md similarity index 64% rename from packages/harbor-manager/README.md rename to packages/metaharness/README.md index f83b3c54d..5cfd75a8f 100644 --- a/packages/harbor-manager/README.md +++ b/packages/metaharness/README.md @@ -1,4 +1,4 @@ -# @oh-my-pi/harbor-manager +# @oh-my-pi/pi-metaharness One manager for repository benchmarks. Harbor, TypeScript edit, and SnapCompact runs use the same experiment → run → trace model, SQLite store, REST/SSE API, @@ -30,8 +30,19 @@ bun run serve --port 4700 ## Server - `GET /` — experiments, runs, normalized traces, and a launch form for every benchmark. -- `GET /api/experiments` — experiment summaries across all benchmark types. -- `GET /api/runs` — uniform run rows with benchmark, score, progress, spend, and tokens. +- `GET /api/experiments[?q=]` — experiment summaries across all benchmark types + (`q` filters by id/goal substring). +- `POST /api/experiments` — register an experiment before its first arm. Body + `{ "id": "sb2", "goal": "..." }`; the id is the dash-free token job names + group under (`sb2-n8` → experiment `sb2`). +- `GET /api/experiments/:id` — arms, per-task matrix, and calibrated projections. +- `PUT /api/experiments/:id` — update the goal and per-run role/note/label. +- `POST /api/experiments/:id/arms` — launch a comparable arm; sample + config + inherited from a sibling. +- `DELETE /api/experiments/:id` — delete every arm (DB rows **and** job dirs) + plus the goal row; rejected while any arm is running. +- `GET /api/runs[?experiment=&status=&benchmark=]` — uniform run rows with + benchmark, score, progress, spend, and tokens. - `POST /api/runs` — launch through a benchmark adapter. Body: ```json @@ -48,14 +59,24 @@ bun run serve --port 4700 ``` `benchmark` is `harbor`, `edit`, or `snapcompact`. Harbor uses `dataset`, - `include`, `timeoutMultiplier`, and `downshift`; edit uses `include` as task IDs; + `include`, `timeoutMultiplier`, and `prewalk`; edit uses `include` as task IDs; SnapCompact uses `conditions` and treats `tasks` as the passage limit. - `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read). -- `DELETE /api/runs/:name` — cancel a manager-launched run. +- `POST /api/runs/:name/cancel` — cancel a manager-launched run. +- `DELETE /api/runs/:name` — permanently delete a finished run (DB row **and** + job dir; a surviving dir would be re-discovered on restart); rejected while + the run is live. +- `POST /api/runs/:name/resume` — resume an incomplete harbor run in place: + completed trials (and their spend) are reused, interrupted/pending trials + re-run, and errored trials retried (body `{ "filterErrorTypes": [...] }` + overrides the retry set, which defaults to every exception type in the job's + `result.json`). The runner recovers the original launch flags from + `_bench//runner-config.json` (snapshotted at launch) or the run's + `manager.json` — nothing needs re-specifying. - `GET /api/runs/:name/traces/:trace[?raw=1]` — normalized or native trace. - `GET /api/events` — SSE stream of run-list snapshots (sent on change). -State lives in `/_manager/harbor-manager.sqlite`; the filesystem +State lives in `/_manager/metaharness.sqlite`; the filesystem stays the source of truth and historical CLI runs are auto-discovered. ## Harbor runner options (excerpt) @@ -77,6 +98,8 @@ stays the source of truth and historical CLI runs are auto-discovered. | `--gateway-url ` | `http://host.docker.internal:4000` | `http://192.168.64.1:4000` under `--environment apple-container` | | `--no-gateway` | off | Pass host provider keys into containers instead | | `-o, --jobs-dir ` | `/runs/harbor` | Shared with the server | +| `--resume ` | — | Resume that job dir via `harbor job resume`; original flags recovered automatically | +| `--filter-error-type ` | `CancelledError` | With `--resume`: also re-run completed trials that errored with exception type `T` (repeatable) | | `--dry-run` | off | Print the harbor command + models.yml and exit | ## Outputs @@ -86,6 +109,24 @@ stays the source of truth and historical CLI runs are auto-discovered. - `/_bench//harbor.log` — full Harbor output. - `/_manager/logs/.log` — runner output for API-launched runs. +## Trace reports + +`scripts/trace-report.ts` turns one run trace into a narrative markdown report +(numbered Turn Log with one grounded sentence per assistant turn, harness +notices in place, then a Story Arc and — for failed runs — a failure analysis). +It map/reduces the normalized trace through two cheap OpenRouter models +(defaults: `inclusionai/ling-2.6-flash` per turn, `openai/gpt-oss-120b` for the +arc; ~$0.001 per report). API keys resolve through omp's auth storage. + +```bash +bun scripts/trace-report.ts [--focus "reviewer notes"] [--out report.md] +bun scripts/trace-report.ts "sb3-ntg|django__django-12325__ddQroP4" # run|trace also accepted +``` + +Flags: `--base` (server, default `http://localhost:4700`), `--tiny` / `--synth` +(`/` overrides), `--focus` (extra reviewer context, e.g. the +known-correct fix for a failed task), `--concurrency` (default 8). + ## Caveats - **Network policy.** On Harbor's local Docker backend only **public** diff --git a/packages/harbor-manager/adapters/edit/bun-imports.d.ts b/packages/metaharness/adapters/edit/bun-imports.d.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/bun-imports.d.ts rename to packages/metaharness/adapters/edit/bun-imports.d.ts diff --git a/packages/harbor-manager/adapters/edit/cli.ts b/packages/metaharness/adapters/edit/cli.ts similarity index 98% rename from packages/harbor-manager/adapters/edit/cli.ts rename to packages/metaharness/adapters/edit/cli.ts index dbbe8ce04..eff9e91de 100644 --- a/packages/harbor-manager/adapters/edit/cli.ts +++ b/packages/metaharness/adapters/edit/cli.ts @@ -11,7 +11,7 @@ import { type BenchmarkConfig, runBenchmark } from "./runner"; const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark"); async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> { - const temp = await TempDir.create("@harbor-edit-fixtures-"); + const temp = await TempDir.create("@metaharness-edit-fixtures-"); const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer()); for (const [filePath, file] of await archive.files()) { await Bun.write(path.join(temp.path(), filePath), file); diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md b/packages/metaharness/adapters/edit/prompts/benchmark-retry.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md rename to packages/metaharness/adapters/edit/prompts/benchmark-retry.md diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-system.md b/packages/metaharness/adapters/edit/prompts/benchmark-system.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-system.md rename to packages/metaharness/adapters/edit/prompts/benchmark-system.md diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-task.md b/packages/metaharness/adapters/edit/prompts/benchmark-task.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-task.md rename to packages/metaharness/adapters/edit/prompts/benchmark-task.md diff --git a/packages/harbor-manager/adapters/edit/report.ts b/packages/metaharness/adapters/edit/report.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/report.ts rename to packages/metaharness/adapters/edit/report.ts diff --git a/packages/harbor-manager/adapters/edit/runner.test.ts b/packages/metaharness/adapters/edit/runner.test.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/runner.test.ts rename to packages/metaharness/adapters/edit/runner.test.ts diff --git a/packages/harbor-manager/adapters/edit/runner.ts b/packages/metaharness/adapters/edit/runner.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/runner.ts rename to packages/metaharness/adapters/edit/runner.ts diff --git a/packages/harbor-manager/adapters/edit/tsconfig.json b/packages/metaharness/adapters/edit/tsconfig.json similarity index 100% rename from packages/harbor-manager/adapters/edit/tsconfig.json rename to packages/metaharness/adapters/edit/tsconfig.json diff --git a/packages/harbor-manager/agent/omp_local.py b/packages/metaharness/agent/omp_local.py similarity index 97% rename from packages/harbor-manager/agent/omp_local.py rename to packages/metaharness/agent/omp_local.py index 7cb9f32be..46c4fb9d0 100644 --- a/packages/harbor-manager/agent/omp_local.py +++ b/packages/metaharness/agent/omp_local.py @@ -433,7 +433,7 @@ class OmpLocal(BaseInstalledAgent): ) def _generate_models_yaml(self) -> str: - lines = ["# Generated by harbor-manager runner — routes auth via host gateway.", "providers:"] + lines = ["# Generated by metaharness runner — routes auth via host gateway.", "providers:"] for provider in self._gateway_providers: lines += [ f" {provider}:", @@ -450,7 +450,7 @@ class OmpLocal(BaseInstalledAgent): web_search can't authenticate through the gateway, so it's off by default. """ lines = [ - "# Generated by harbor-manager runner.", + "# Generated by metaharness runner.", "web_search:", f" enabled: {'true' if self._web_search else 'false'}", ] @@ -565,13 +565,18 @@ class OmpLocal(BaseInstalledAgent): } def _sum_main(self, path: Path, acc: "_Usage") -> None: - """Sum assistant `message_end` usage from omp's stdout JSONL.""" + """Sum assistant `message_end` usage from omp's stdout JSONL. + + Streams line-by-line: a runaway transcript must not OOM the host-side + post-run parse. + """ if not path.exists(): return - for line in path.read_text(errors="replace").splitlines(): - event = _loads(line) - if not event or event.get("type") != "message_end": - continue - message = event.get("message") - if isinstance(message, dict) and message.get("role") == "assistant": - acc.add(message.get("usage")) + with path.open(errors="replace") as fh: + for line in fh: + event = _loads(line) + if not event or event.get("type") != "message_end": + continue + message = event.get("message") + if isinstance(message, dict) and message.get("role") == "assistant": + acc.add(message.get("usage")) diff --git a/packages/harbor-manager/package.json b/packages/metaharness/package.json similarity index 81% rename from packages/harbor-manager/package.json rename to packages/metaharness/package.json index d6e7af6e7..f26449d10 100644 --- a/packages/harbor-manager/package.json +++ b/packages/metaharness/package.json @@ -1,7 +1,7 @@ { "type": "module", "private": true, - "name": "@oh-my-pi/harbor-manager", + "name": "@oh-my-pi/pi-metaharness", "version": "0.0.1", "description": "Unified benchmark runners plus Harbor run storage, REST/SSE APIs, and a live web dashboard", "homepage": "https://omp.sh", @@ -10,24 +10,25 @@ "repository": { "type": "git", "url": "git+https://github.com/can1357/oh-my-pi.git", - "directory": "packages/harbor-manager" + "directory": "packages/metaharness" }, "bin": { - "harbor-manager": "src/server.ts" + "metaharness": "src/server.ts" }, "scripts": { "check": "biome check . && bun run check:types", - "check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit", + "check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit && tsgo -p scripts/tsconfig.json --noEmit", "lint": "biome lint .", "start": "bun run src/server.ts", "serve": "bun run src/server.ts", - "dev": "bun scripts/dev.ts", + "dev": "bun --hot src/server.ts", "test": "bun test" }, "dependencies": { "@oh-my-pi/hashline": "catalog:", "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "@oh-my-pi/typescript-edit-benchmark": "workspace:*", @@ -46,8 +47,7 @@ "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", "@types/react-dom": "^19.1.0", - "@vitejs/plugin-react": "^5.0.4", - "vite": "catalog:" + "react-refresh": "^0.18.0" }, "engines": { "bun": ">=1.3.14" diff --git a/packages/metaharness/scripts/trace-report.ts b/packages/metaharness/scripts/trace-report.ts new file mode 100755 index 000000000..049b75d55 --- /dev/null +++ b/packages/metaharness/scripts/trace-report.ts @@ -0,0 +1,397 @@ +#!/usr/bin/env bun +/** + * Narrative trace report for a metaharness run trace. + * + * Two-stage map/reduce over the normalized trace JSON served by the + * metaharness server (`GET /api/runs/:run/traces/:trace`): + * + * 1. Map — every assistant turn (its prose, tool calls, and full tool + * result bodies) is handed to a very cheap "tiny" model which returns a + * single grounded sentence describing what the agent did and what the + * results showed. Turns are independent, so this fans out in parallel. + * 2. Reduce — the deterministic numbered Turn Log (tool names come from the + * trace itself, only the grounded sentence is model-written) plus run + * metadata, harness notices, error excerpts, and the final assistant + * prose go to a slightly smarter (still cheap) model which writes the + * Story Arc and, for failed runs, the failure analysis. + * + * Usage: + * bun scripts/trace-report.ts + * bun scripts/trace-report.ts "|" # or run/trace + * ... --focus "known-correct fix is X; compare" # reviewer notes + * ... --out report.md + * ... --tiny openrouter/inclusionai/ling-2.6-flash + * ... --synth openrouter/openai/gpt-oss-120b + * + * Auth: provider API keys resolve through omp's auth storage + * (~/.omp/agent/agent.db: stored key, OAuth, or env var fallback). + */ + +import { parseArgs } from "node:util"; +import { type Api, AuthStorage, completeSimple, type Model, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai"; +import { type GeneratedProvider, getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { getAgentDbPath } from "@oh-my-pi/pi-utils"; + +const DEFAULT_TINY = "openrouter/inclusionai/ling-2.6-flash"; +const DEFAULT_SYNTH = "openrouter/openai/gpt-oss-120b"; +const DEFAULT_BASE = "http://localhost:4700"; + +// -------------------------------------------------------------------------- +// Trace API types (mirror packages/metaharness/src/store.ts normalization) + +interface TraceAssistantEntry { + kind: "assistant"; + model: string; + text: string; + tools: string[]; +} + +interface TraceToolResultEntry { + kind: "toolResult"; + tool: string; + isError: boolean; + text: string; +} + +interface TraceNoticeEntry { + kind: "notice"; + text: string; +} + +type TraceEntry = TraceAssistantEntry | TraceToolResultEntry | TraceNoticeEntry; + +interface TraceResponse { + jobName: string; + trace: string; + entries: TraceEntry[]; + totalEvents: number; +} + +interface RunTraceRow { + name: string; + task: string; + status: string; + reward: number | null; + costUsd: number | null; + durationMs: number | null; +} + +interface RunResponse { + run: { benchmark: string; dataset: string; models: string; jobName: string }; + traces: RunTraceRow[]; +} + +// -------------------------------------------------------------------------- +// Turn grouping + +/** One numbered item of the Turn Log: an assistant turn or a harness notice. */ +type LogItem = + | { kind: "turn"; model: string; text: string; tools: string[]; results: TraceToolResultEntry[] } + | { kind: "notice"; text: string }; + +function groupItems(entries: TraceEntry[]): LogItem[] { + const items: LogItem[] = []; + let current: Extract | undefined; + for (const entry of entries) { + if (entry.kind === "assistant") { + current = { kind: "turn", model: entry.model, text: entry.text, tools: entry.tools, results: [] }; + items.push(current); + } else if (entry.kind === "toolResult") { + if (!current) throw new Error("trace starts with a toolResult; cannot attach it to a turn"); + current.results.push(entry); + } else { + items.push({ kind: "notice", text: entry.text }); + } + } + return items; +} + +// -------------------------------------------------------------------------- +// Model + auth + +interface OpenedModel { + model: Model; + apiKey: string; + spec: string; + usage: { input: number; output: number; calls: number }; +} + +async function openModel(modelSpec: string, storage: AuthStorage): Promise { + const slash = modelSpec.indexOf("/"); + if (slash <= 0) throw new Error(`model must be /, got "${modelSpec}"`); + const provider = modelSpec.slice(0, slash); + const modelId = modelSpec.slice(slash + 1); + const model = getBundledModel(provider as GeneratedProvider, modelId); + if (!model) throw new Error(`unknown model "${modelSpec}" (not in bundled catalog)`); + const apiKey = await storage.getApiKey(provider); + if (!apiKey) { + throw new Error(`no credentials for provider "${provider}" (run \`omp login\` or set the provider env var)`); + } + return { model, apiKey, spec: modelSpec, usage: { input: 0, output: 0, calls: 0 } }; +} + +/** One retried oneshot text completion. Throws after `attempts` failures. */ +async function ask(opened: OpenedModel, system: string, user: string, maxTokens: number): Promise { + let lastError = ""; + for (let attempt = 0; attempt < 4; attempt++) { + const response = await completeSimple( + opened.model, + { + systemPrompt: [system], + messages: [{ role: "user", content: [{ type: "text", text: user }], timestamp: Date.now() }], + }, + { apiKey: opened.apiKey, temperature: 0, maxTokens }, + ); + opened.usage.calls++; + opened.usage.input += response.usage.input + response.usage.cacheRead; + opened.usage.output += response.usage.output; + if (response.stopReason === "error" || response.stopReason === "aborted") { + lastError = response.errorMessage ?? response.stopReason; + await Bun.sleep(1000 * (attempt + 1)); + continue; + } + const text = response.content + .filter(content => content.type === "text") + .map(content => content.text) + .join("") + .trim(); + if (text) return text; + lastError = "model returned no text"; + } + throw new Error(`completion failed on ${opened.spec}: ${lastError}`); +} + +// -------------------------------------------------------------------------- +// Map phase: one grounded sentence per assistant turn + +const TINY_SYSTEM = `You annotate one turn of an AI coding-agent transcript. +Reply with exactly ONE sentence (at most 35 words). Plain text only: no markdown, no bullet, no quotes around the whole reply, no preamble. Write in third person ("The agent …"). +The sentence states what the agent did this turn and what the tool results showed. +Be concrete: copy exact file paths, line numbers, function names, commands, test counts, and error messages from the material given. +Describe ONLY what this turn's tool results prove. Never claim something ran, passed, or was fixed unless a result in THIS turn shows it. +The assistant prose states the agent's intent; only tool results are evidence — never present intentions as completed actions. +Never work the agent model name into the sentence. +A todo result is a checklist snapshot: describe the checklist state (items added/completed), never narrate its items as performed actions. +A write result only proves the file was written (path and size). +An edit result shows the affected file lines AFTER the change ([path#TAG] header plus numbered lines): quote the resulting logic precisely; never guess what was removed. +Tags like #C32D in [path#TAG] headers are content hashes, not line numbers — never cite them. +If a tool result is an error, the sentence MUST name the error. +If there are no tool calls, summarize what the assistant prose states or concludes.`; + +const RESULT_EXCERPT_LIMIT = 1400; + +function turnPrompt(turn: Extract): string { + const parts: string[] = [`Agent model: ${turn.model}`]; + parts.push(`Assistant prose: ${turn.text.trim() ? turn.text.trim().slice(0, 2000) : "(none)"}`); + if (turn.results.length === 0) { + parts.push("Tool calls: none."); + } + turn.results.forEach((result, index) => { + const body = + result.text.length > RESULT_EXCERPT_LIMIT ? `${result.text.slice(0, RESULT_EXCERPT_LIMIT)}…` : result.text; + parts.push(`Tool call ${index + 1}: ${result.tool} → ${result.isError ? "ERROR" : "ok"}\n${body}`); + }); + if (turn.results.length > 0 && turn.results.every(result => result.tool === "todo")) { + parts.push( + "NOTE: this turn only updated the todo checklist. The checklist items are PLANS, not events; your sentence must summarize only the checklist state (item counts, statuses, the in-progress item).", + ); + } + parts.push("One sentence:"); + return parts.join("\n\n"); +} + +/** Map `items` through `worker` with at most `limit` in flight, order preserved. */ +async function mapPool(items: T[], limit: number, worker: (item: T, index: number) => Promise): Promise { + const results = new Array(items.length); + let next = 0; + const lanes = Array.from({ length: Math.min(limit, items.length) }, async () => { + while (next < items.length) { + const index = next++; + results[index] = await worker(items[index], index); + } + }); + await Promise.all(lanes); + return results; +} + +// -------------------------------------------------------------------------- +// Turn Log assembly (deterministic scaffolding + tiny sentences) + +function toolsLine(tools: string[]): string { + if (tools.length === 0) return "prose only (no tool calls)"; + const counts = new Map(); + for (const tool of tools) counts.set(tool, (counts.get(tool) ?? 0) + 1); + const named = [...counts.entries()].map(([tool, n]) => (n > 1 ? `\`${tool}\` ×${n}` : `\`${tool}\``)); + return `tools called: ${named.join(", ")}`; +} + +function renderTurnLog(items: LogItem[], sentences: (string | undefined)[]): string { + const lines: string[] = ["### Turn Log", ""]; + items.forEach((item, index) => { + const number = index + 1; + if (item.kind === "notice") { + lines.push(`${number}. **— harness: notice: "${item.text}"**`); + return; + } + const errored = item.results.filter(result => result.isError).map(result => `\`${result.tool}\``); + const errorNote = errored.length > 0 ? ` (errored: ${errored.join(", ")})` : ""; + lines.push(`${number}. **[${item.model}]** ${toolsLine(item.tools)}${errorNote}.`); + lines.push(` - Grounded action: ${sentences[index] ?? "(summary unavailable)"}`); + }); + return lines.join("\n"); +} + +// -------------------------------------------------------------------------- +// Reduce phase: story arc + failure analysis + +const SYNTH_SYSTEM = `You write the "Story Arc" section of a trace-analysis report for one AI coding-agent benchmark run. +You are given run metadata, a numbered Turn Log (already final — never rewrite or renumber it), the run's final assistant message, and optional reviewer focus notes. +Output ONLY a markdown "### Story Arc" section. +Rules: +- Bullets of the form: - **Turns A–B (N turns): Title (model)**: 1–3 sentence description of that phase. +- The ranges must cover every numbered Turn Log item exactly once, in order, with no gaps or overlaps; harness notices belong to the range containing them and phase boundaries should align with model switches and notices where sensible. +- Every claim must be grounded in the Turn Log or the final assistant message; never invent files, tests, or events. +- If the run status is "fail", end with a final bullet - **Failure analysis**: explaining what the agent actually changed, why the run still failed, and — when reviewer focus notes describe the known-correct fix — how the agent's change diverges from it and what verification would have caught the gap.`; + +function synthPrompt(options: { + run: string; + trace: string; + meta: string; + status: string | undefined; + focus: string | undefined; + turnLog: string; + finalProse: string; +}): string { + const failed = options.status === "fail"; + const parts = [ + `Run: ${options.run}\nTrace: ${options.trace}\n${options.meta}`, + options.focus ? `Reviewer focus notes:\n${options.focus}` : "", + options.turnLog, + `Final assistant message (verbatim, may be truncated):\n"""\n${options.finalProse.slice(0, 4000) || "(none)"}\n"""`, + failed + ? 'Run status is "fail". Write the Story Arc now; the LAST bullet MUST be **Failure analysis** per the rules.' + : "Write the Story Arc now.", + ]; + return parts.filter(Boolean).join("\n\n"); +} + +// -------------------------------------------------------------------------- +// Run + +function formatDuration(ms: number | null): string { + if (ms == null) return "?"; + const seconds = Math.round(ms / 1000); + return `${Math.floor(seconds / 60)}m${String(seconds % 60).padStart(2, "0")}s`; +} + +function usageLine(opened: OpenedModel): string { + const cost = opened.model.cost + ? (opened.usage.input * opened.model.cost.input + opened.usage.output * opened.model.cost.output) / 1e6 + : undefined; + const costText = cost === undefined ? "" : ` ≈ $${cost.toFixed(4)}`; + return `${opened.spec}: ${opened.usage.calls} calls, ${opened.usage.input} in / ${opened.usage.output} out tokens${costText}`; +} + +async function main(): Promise { + const { values, positionals } = parseArgs({ + args: Bun.argv.slice(2), + allowPositionals: true, + options: { + base: { type: "string", default: DEFAULT_BASE }, + tiny: { type: "string", default: DEFAULT_TINY }, + synth: { type: "string", default: DEFAULT_SYNTH }, + focus: { type: "string" }, + out: { type: "string" }, + concurrency: { type: "string", default: "8" }, + }, + }); + + const joined = positionals.join(" ").trim(); + const match = joined.match(/^(\S+?)[|/\s]+(\S+)$/); + if (!match) { + console.error('usage: bun scripts/trace-report.ts (or "|")'); + process.exit(2); + } + const [, run, trace] = match; + + const traceResponse = await fetch(`${values.base}/api/runs/${run}/traces/${trace}`); + if (!traceResponse.ok) + throw new Error(`trace fetch failed: HTTP ${traceResponse.status} ${await traceResponse.text()}`); + const traceData = (await traceResponse.json()) as TraceResponse; + + let meta = ""; + let status: string | undefined; + try { + const runResponse = await fetch(`${values.base}/api/runs/${run}`); + if (runResponse.ok) { + const runData = (await runResponse.json()) as RunResponse; + const row = runData.traces.find(candidate => candidate.name === trace); + status = row?.status; + meta = [ + `Benchmark: ${runData.run.benchmark} (${runData.run.dataset})`, + `Configured model: ${runData.run.models}`, + row + ? `Task: ${row.task} — status: ${row.status.toUpperCase()} (reward ${row.reward ?? "?"}), cost $${row.costUsd?.toFixed(2) ?? "?"}, duration ${formatDuration(row.durationMs)}` + : "", + ] + .filter(Boolean) + .join("\n"); + } + } catch { + // Report still works from the trace alone. + } + + const items = groupItems(traceData.entries); + const turnCount = items.filter(item => item.kind === "turn").length; + console.error(`[trace-report] ${items.length} log items (${turnCount} turns) from ${traceData.totalEvents} events`); + + const store = await SqliteAuthCredentialStore.open(getAgentDbPath()); + const storage = new AuthStorage(store); + await storage.reload(); + const tiny = await openModel(values.tiny, storage); + const synth = values.synth === values.tiny ? tiny : await openModel(values.synth, storage); + + // Map: one grounded sentence per assistant turn. + let completed = 0; + const sentences = await mapPool(items, Number(values.concurrency), async item => { + if (item.kind !== "turn") return undefined; + try { + const sentence = await ask(tiny, TINY_SYSTEM, turnPrompt(item), 300); + return sentence.replace(/\s+/g, " ").trim(); + } finally { + completed++; + if (completed % 10 === 0) console.error(`[trace-report] map ${completed}/${items.length}`); + } + }); + + const turnLog = renderTurnLog(items, sentences); + + // Reduce: story arc + failure analysis. + const finalTurn = [...items].reverse().find(item => item.kind === "turn" && item.text.trim()); + const finalProse = finalTurn?.kind === "turn" ? finalTurn.text : ""; + const storyArc = await ask( + synth, + SYNTH_SYSTEM, + synthPrompt({ run, trace, meta, status, focus: values.focus, turnLog, finalProse }), + 3000, + ); + + const report = [ + `# Trace report: ${run} / ${trace}`, + meta, + turnLog, + storyArc.trim(), + `---\n_${usageLine(tiny)}${synth === tiny ? "" : `; ${usageLine(synth)}`}_`, + ] + .filter(Boolean) + .join("\n\n"); + + if (values.out) { + await Bun.write(values.out, `${report}\n`); + console.error(`[trace-report] wrote ${values.out}`); + } else { + console.log(report); + } +} + +await main(); diff --git a/packages/metaharness/scripts/tsconfig.json b/packages/metaharness/scripts/tsconfig.json new file mode 100644 index 000000000..f45e6034a --- /dev/null +++ b/packages/metaharness/scripts/tsconfig.json @@ -0,0 +1,4 @@ +{ + "extends": "../../tsconfig.workspace.json", + "include": ["."] +} diff --git a/packages/harbor-manager/src/adapters/snapcompact.py b/packages/metaharness/src/adapters/snapcompact.py similarity index 100% rename from packages/harbor-manager/src/adapters/snapcompact.py rename to packages/metaharness/src/adapters/snapcompact.py diff --git a/packages/harbor-manager/src/benchmarks.test.ts b/packages/metaharness/src/benchmarks.test.ts similarity index 100% rename from packages/harbor-manager/src/benchmarks.test.ts rename to packages/metaharness/src/benchmarks.test.ts diff --git a/packages/harbor-manager/src/benchmarks.ts b/packages/metaharness/src/benchmarks.ts similarity index 100% rename from packages/harbor-manager/src/benchmarks.ts rename to packages/metaharness/src/benchmarks.ts diff --git a/packages/harbor-manager/src/experiments.test.ts b/packages/metaharness/src/experiments.test.ts similarity index 96% rename from packages/harbor-manager/src/experiments.test.ts rename to packages/metaharness/src/experiments.test.ts index cd0e8f630..053f068cc 100644 --- a/packages/harbor-manager/src/experiments.test.ts +++ b/packages/metaharness/src/experiments.test.ts @@ -24,7 +24,7 @@ function runRow(overrides: Partial): RunRow { agent: "omp", models: "anthropic/claude-opus-4-8", label: "", - downshift: null, + prewalk: null, config: {}, role: "", note: "", @@ -125,11 +125,11 @@ describe("summarizeArm", () => { expect(finished.costPerTask).toBeCloseTo(1, 5); }); - it("describes the downshift config in the arm line", () => { + it("describes the prewalk config in the arm line", () => { const arm = summarizeArm( runRow({ jobName: "sb2-nact", - downshift: JSON.stringify({ into: "google/gemini-3.5-flash" }), + prewalk: JSON.stringify({ into: "google/gemini-3.5-flash" }), }), [], ); @@ -140,7 +140,7 @@ describe("summarizeArm", () => { const arm = summarizeArm( runRow({ jobName: "sb2-nact", - downshift: JSON.stringify({ model: "google/gemini-3.5-flash", onAction: true, plan: true }), + prewalk: JSON.stringify({ model: "google/gemini-3.5-flash", onAction: true, plan: true }), }), [], ); diff --git a/packages/harbor-manager/src/experiments.ts b/packages/metaharness/src/experiments.ts similarity index 92% rename from packages/harbor-manager/src/experiments.ts rename to packages/metaharness/src/experiments.ts index 9f19c97ef..5d5a59b1a 100644 --- a/packages/harbor-manager/src/experiments.ts +++ b/packages/metaharness/src/experiments.ts @@ -19,7 +19,7 @@ export interface ArmSummary { run: RunRow; /** Arm label: job name minus the experiment prefix. */ arm: string; - /** Human config line: models plus downshift description when known. */ + /** Human config line: models plus prewalk description when known. */ config: string; /** Observed pass% over decided trials. */ passPct: number | null; @@ -67,11 +67,11 @@ export function armOf(jobName: string): string { return jobName.length > exp.length ? jobName.slice(exp.length + 1) : jobName; } -function downshiftLabel(downshiftJson: string | null): string { - if (!downshiftJson) return ""; +function prewalkLabel(prewalkJson: string | null): string { + if (!prewalkJson) return ""; try { // Historical rows may hold legacy reasoning-slide JSON ({model, turns, onAction, plan}). - const parsed = JSON.parse(downshiftJson) as { + const parsed = JSON.parse(prewalkJson) as { into?: string; model?: string; turns?: number; @@ -118,7 +118,7 @@ export function summarizeArm(run: RunRow, traces: TraceRow[]): ArmSummary { return { run, arm: armOf(run.jobName), - config: `${run.benchmark} · ${run.models}${downshiftLabel(run.downshift)}`, + config: `${run.benchmark} · ${run.models}${prewalkLabel(run.prewalk)}`, passPct, costPerTask, meanTrialMs, @@ -204,7 +204,7 @@ export function buildExperiments(store: RunStore): ExperimentSummary[] { for (const [id, runs] of groups) { out.push({ id, - goal: store.getExperimentGoal(id), + goal: store.getExperimentMeta(id)?.goal ?? "", arms: runs.length, runningArms: runs.filter(r => r.status === "running").length, datasets: [...new Set(runs.map(r => r.dataset).filter(Boolean))], @@ -218,6 +218,26 @@ export function buildExperiments(store: RunStore): ExperimentSummary[] { updatedAt: Math.max(...runs.map(r => r.finishedAt ?? Date.now())), }); } + // Registered-but-empty experiments (created via POST /api/experiments, no + // arms yet) are still browsable: zeroed rollups, goal from the meta row. + for (const meta of store.listExperimentMeta()) { + if (groups.has(meta.id)) continue; + out.push({ + id: meta.id, + goal: meta.goal, + arms: 0, + runningArms: 0, + datasets: [], + nTotal: 0, + done: 0, + pass: 0, + fail: 0, + error: 0, + costUsd: 0, + createdAt: meta.updatedAt, + updatedAt: meta.updatedAt, + }); + } out.sort((a, b) => b.updatedAt - a.updatedAt); return out; } @@ -259,7 +279,11 @@ export function pickMergedTrials(traces: TraceRow[]): TraceRow[] { export function experimentDetail(store: RunStore, id: string): ExperimentDetail | null { const runs = store.listRuns().filter(r => experimentOf(r.jobName) === id); - if (runs.length === 0) return null; + if (runs.length === 0) { + // Registered but armless (POST /api/experiments): still readable. + const meta = store.getExperimentMeta(id); + return meta ? { id, goal: meta.goal, arms: [], tasks: [], matrix: {} } : null; + } // One row per CANONICAL arm: `-fix`/`-backfill` re-runs merge into their // base arm — per-task best trial, summed spend. const groups = new Map(); @@ -347,5 +371,5 @@ export function experimentDetail(store: RunStore, id: string): ExperimentDetail // "reference rows, then treatments". const roleRank = (role: string) => (role === "baseline" ? 0 : role === "variant" ? 1 : 2); arms.sort((a, b) => roleRank(a.run.role) - roleRank(b.run.role) || a.arm.localeCompare(b.arm)); - return { id, goal: store.getExperimentGoal(id), arms, tasks: [...tasks].sort(), matrix }; + return { id, goal: store.getExperimentMeta(id)?.goal ?? "", arms, tasks: [...tasks].sort(), matrix }; } diff --git a/packages/metaharness/src/launch-args.ts b/packages/metaharness/src/launch-args.ts new file mode 100644 index 000000000..b1afc9c65 --- /dev/null +++ b/packages/metaharness/src/launch-args.ts @@ -0,0 +1,84 @@ +/** + * Shared launch surface: the POST /api/runs request shape and its mapping to + * runner CLI argv. The server uses it to spawn new harbor runs; the runner's + * `--resume` uses it to rebuild the original invocation from a job dir's + * manager.json launch record when no runner-config.json snapshot exists. + */ +import * as fs from "node:fs"; +import * as path from "node:path"; +import type { BenchmarkKind, RunRole } from "./store"; + +const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); + +/** POST /api/runs body. Mirrors the runner CLI surface we actually use. */ +export interface LaunchRequest { + /** Benchmark adapter to execute. */ + benchmark?: BenchmarkKind; + model: string; + dataset?: string; + /** Task count for a dataset sample, or omit when `include` is given. */ + tasks?: number; + /** Explicit task names (passed as repeated --include). */ + include?: string[]; + concurrency?: number; + /** SnapCompact conditions; ignored by other benchmarks. */ + conditions?: string[]; + timeoutMultiplier?: number; + attempts?: number; + agent?: string; + jobName?: string; + webSearch?: boolean; + /** Harbor container backend. Defaults to apple-container whenever the Apple `container` CLI is installed; docker is an explicit opt-in. */ + environment?: "docker" | "apple-container"; + /** Prewalk to a fast/cheap model at the first edit/write once the todo list exists; `into` overrides the default "smol" target. */ + prewalk?: { into?: string }; + /** Role of this run inside its experiment (baseline vs treatment). */ + role?: RunRole; + /** One-line description of what this arm tests. */ + note?: string; + /** Experiment goal; upserted for the run's experiment (job-name prefix). */ + goal?: string; + /** Use prebuilt dist/omp-linux-* binaries instead of the default source mount. */ + prebuiltBinaries?: boolean; + /** Extra raw runner args, appended verbatim. */ + extraArgs?: string[]; +} + +/** Runner CLI flags (sans the `bun src/runner.ts` prefix) for a harbor launch. */ +export function harborRunnerArgs( + request: LaunchRequest, + opts: { jobsDir: string; jobName: string; dataset: string }, +): string[] { + const argv = ["--model", request.model, "-d", opts.dataset, "--job-name", opts.jobName, "--jobs-dir", opts.jobsDir]; + // Prefer Apple Container when its CLI is present: native arm64 task + // containers with no Docker daemon. The runner itself defaults to + // docker, so the preference must be stated here. + const environment = request.environment ?? (Bun.which("container") ? "apple-container" : "docker"); + argv.push("--environment", environment); + if (request.agent) argv.push("--agent", request.agent); + // An explicit include list IS the sample — never let the runner's + // default task cap truncate it. + const tasks = request.tasks ?? (request.include && request.include.length > 0 ? request.include.length : undefined); + if (tasks !== undefined) argv.push("--tasks", String(tasks)); + if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency)); + if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts)); + if (request.timeoutMultiplier !== undefined) argv.push("--timeout-multiplier", String(request.timeoutMultiplier)); + if (request.webSearch) argv.push("--web-search"); + for (const task of request.include ?? []) argv.push("--include", task); + if (request.prewalk) { + argv.push("--agent-arg", "--prewalk"); + if (request.prewalk.into) { + argv.push("--agent-arg", "--prewalk-into", "--agent-arg", request.prewalk.into); + const provider = request.prewalk.into.split("/", 1)[0]; + if (provider && request.prewalk.into.includes("/")) argv.push("--providers", provider); + } + } + if (request.prebuiltBinaries) { + for (const name of ["omp-linux-arm64", "omp-linux-x64"]) { + const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name); + if (fs.existsSync(binary)) argv.push("--binary", binary); + } + } + argv.push(...(request.extraArgs ?? [])); + return argv; +} diff --git a/packages/harbor-manager/src/manager.test.ts b/packages/metaharness/src/manager.test.ts similarity index 66% rename from packages/harbor-manager/src/manager.test.ts rename to packages/metaharness/src/manager.test.ts index 42e6f3224..70044461e 100644 --- a/packages/harbor-manager/src/manager.test.ts +++ b/packages/metaharness/src/manager.test.ts @@ -19,7 +19,7 @@ afterEach(() => { }); function makeJobsDir(): string { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-manager-test-")); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "metaharness-test-")); cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); return dir; } @@ -145,9 +145,7 @@ describe("RunStore", () => { store.discover(); store.setExperimentGoal("exp", "does the treatment beat the baseline?"); expect(store.setRunMeta("exp-base", { role: "baseline", note: "plain model" })).toBe(true); - expect(store.setRunMeta("exp-treat", { role: "variant", note: "downshift flash", label: "flash@edit" })).toBe( - true, - ); + expect(store.setRunMeta("exp-treat", { role: "variant", note: "prewalk flash", label: "flash@edit" })).toBe(true); expect(store.setRunMeta("exp-missing", { role: "variant" })).toBe(false); const detail = experimentDetail(store, "exp"); @@ -155,18 +153,18 @@ describe("RunStore", () => { // ArmSummary.arm resolves to the display label when one is set. expect(detail?.arms.map(a => [a.arm, a.run.role, a.run.note, a.run.label])).toEqual([ ["base", "baseline", "plain model", ""], - ["flash@edit", "variant", "downshift flash", "flash@edit"], + ["flash@edit", "variant", "prewalk flash", "flash@edit"], ]); // Partial updates keep the omitted fields. - expect(store.setRunMeta("exp-treat", { note: "downshift flash v2" })).toBe(true); + expect(store.setRunMeta("exp-treat", { note: "prewalk flash v2" })).toBe(true); const treat = store.getRun("exp-treat"); expect(treat?.label).toBe("flash@edit"); expect(treat?.role).toBe("variant"); - expect(treat?.note).toBe("downshift flash v2"); + expect(treat?.note).toBe("prewalk flash v2"); }); - it("finalizes running rows whose owning process died", () => { + it("releases a dead runner's pid without failing a possibly-live orphan", () => { const jobsDir = makeJobsDir(); writeFixtureJob(jobsDir, "job-b"); const store = new RunStore(jobsDir); @@ -181,7 +179,26 @@ describe("RunStore", () => { }); const rows = store.syncActive(); expect(rows).toHaveLength(1); - expect(store.getRun("job-b")?.status).toBe("failed"); + // The runner is only a monitor: its death must not fail the run while + // the job dir is fresh (an orphaned harbor may still be writing trials). + const row = store.getRun("job-b"); + expect(row?.pid).toBeNull(); + expect(row?.status).toBe("running"); + + // Once harbor stamps the terminal marker, the same sweep completes it. + const jobDir = path.join(jobsDir, "job-b"); + fs.writeFileSync( + path.join(jobDir, "result.json"), + JSON.stringify({ + n_total_trials: 3, + stats: { n_running_trials: 0, n_pending_trials: 0 }, + finished_at: "2026-07-12T11:00:00", + }), + ); + store.syncActive(); + const finished = store.getRun("job-b"); + expect(finished?.status).toBe("complete"); + expect(finished?.finishedAt).toBe(Date.parse("2026-07-12T11:00:00")); }); }); @@ -221,10 +238,13 @@ describe("ManagerServer API", () => { }); expect(badLaunch.status).toBe(400); - const cancelUnknown = (await (await fetch(`${base}/api/runs/nope`, { method: "DELETE" })).json()) as { + const cancelUnknown = (await (await fetch(`${base}/api/runs/nope/cancel`, { method: "POST" })).json()) as { cancelled: boolean; }; expect(cancelUnknown.cancelled).toBe(false); + + const deleteUnknown = await fetch(`${base}/api/runs/nope`, { method: "DELETE" }); + expect(deleteUnknown.status).toBe(404); }); it("serves edit and SnapCompact metrics and native traces through one API", async () => { @@ -302,6 +322,147 @@ describe("ManagerServer API", () => { ).json()) as { entries: Array<{ kind: string }> }; expect(snapTrace.entries.map(entry => entry.kind)).toEqual(["question", "answer", "reference"]); }); + it("guards resume: unknown, non-harbor, running, and config-less runs are rejected", async () => { + const jobsDir = makeJobsDir(); + const manager = new ManagerServer(jobsDir); + manager.store.registerLaunch({ + benchmark: "edit", + jobName: "edit-x", + dataset: "typescript-edit", + agent: "edit", + models: ["m/x"], + pid: process.pid, + }); + manager.store.markExit("edit-x", 1); + // A live harbor run: pid is this test process, never marked exited. + manager.store.registerLaunch({ + benchmark: "harbor", + jobName: "job-live", + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + // A failed harbor run whose job dir has no harbor config.json. + manager.store.registerLaunch({ + benchmark: "harbor", + jobName: "job-bare", + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + manager.store.markExit("job-bare", 1); + const server = manager.start(0); + cleanups.push(() => { + void manager.stop(); + }); + const base = `http://localhost:${server.port}`; + const resumeError = async (name: string): Promise => { + const res = await fetch(`${base}/api/runs/${name}/resume`, { method: "POST" }); + expect(res.status).toBe(400); + return ((await res.json()) as { error: string }).error; + }; + + expect(await resumeError("nope")).toMatch(/not found/); + expect(await resumeError("edit-x")).toMatch(/only harbor/); + expect(await resumeError("job-live")).toMatch(/already running/); + expect(await resumeError("job-bare")).toMatch(/no harbor config.json/); + }); + + it("experiment CRUD: create is browsable, delete removes rows + job dirs, live arms are protected", async () => { + const jobsDir = makeJobsDir(); + const manager = new ManagerServer(jobsDir); + // Two finished arms of experiment `crud` and one live run in a different experiment. + for (const jobName of ["crud-base", "crud-treat"]) { + manager.store.registerLaunch({ + benchmark: "harbor", + jobName, + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + manager.store.markExit(jobName, 0); + } + manager.store.registerLaunch({ + benchmark: "harbor", + jobName: "live-run", + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + const server = manager.start(0); + cleanups.push(() => { + void manager.stop(); + }); + const base = `http://localhost:${server.port}`; + + // Create: registered id is browsable before any run exists. + const created = await fetch(`${base}/api/experiments`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ id: "fresh", goal: "does X beat Y?" }), + }); + expect(created.status).toBe(201); + const list = (await (await fetch(`${base}/api/experiments`)).json()) as Array<{ + id: string; + goal: string; + arms: number; + }>; + const fresh = list.find(e => e.id === "fresh"); + expect(fresh).toMatchObject({ goal: "does X beat Y?", arms: 0 }); + const freshDetail = (await (await fetch(`${base}/api/experiments/fresh`)).json()) as { + goal: string; + arms: unknown[]; + }; + expect(freshDetail).toMatchObject({ goal: "does X beat Y?", arms: [] }); + + // Create: dashed / empty ids can never own a run — rejected. + for (const id of ["bad-id", ""]) { + const res = await fetch(`${base}/api/experiments`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ id }), + }); + expect(res.status).toBe(400); + } + + // Browse: list filters. + const filtered = (await (await fetch(`${base}/api/runs?experiment=crud`)).json()) as Array<{ + jobName: string; + }>; + expect(filtered.map(r => r.jobName).sort()).toEqual(["crud-base", "crud-treat"]); + const running = (await (await fetch(`${base}/api/runs?status=running`)).json()) as Array<{ + jobName: string; + }>; + expect(running.map(r => r.jobName)).toEqual(["live-run"]); + const q = (await (await fetch(`${base}/api/experiments?q=fresh`)).json()) as Array<{ id: string }>; + expect(q.map(e => e.id)).toEqual(["fresh"]); + + // Delete run: live runs are protected, finished runs vanish from DB and disk. + const liveDelete = await fetch(`${base}/api/runs/live-run`, { method: "DELETE" }); + expect(liveDelete.status).toBe(400); + const runDelete = await fetch(`${base}/api/runs/crud-treat`, { method: "DELETE" }); + expect(runDelete.status).toBe(200); + expect(fs.existsSync(path.join(jobsDir, "crud-treat"))).toBe(false); + expect(manager.store.getRun("crud-treat")).toBeNull(); + + // Delete experiment: remaining arm rows + dirs + goal row all go; 404 after. + const expDelete = (await (await fetch(`${base}/api/experiments/crud`, { method: "DELETE" })).json()) as { + deletedRuns: string[]; + }; + expect(expDelete.deletedRuns).toEqual(["crud-base"]); + expect(fs.existsSync(path.join(jobsDir, "crud-base"))).toBe(false); + expect((await fetch(`${base}/api/experiments/crud`)).status).toBe(404); + expect((await fetch(`${base}/api/experiments/unknown`, { method: "DELETE" })).status).toBe(404); + + // Delete experiment with a live arm: refused, nothing removed. + const liveExpDelete = await fetch(`${base}/api/experiments/live`, { method: "DELETE" }); + expect(liveExpDelete.status).toBe(400); + expect(manager.store.getRun("live-run")).not.toBeNull(); + }); }); describe("resolveArmLaunch", () => { @@ -328,8 +489,8 @@ describe("resolveArmLaunch", () => { arm: "n8", model: "google/gemini-3.5-flash", role: "variant", - note: "downshift@flash", - downshift: { into: "google/gemini-3.5-flash" }, + note: "prewalk@flash", + prewalk: { into: "google/gemini-3.5-flash" }, }); expect(launch.jobName).toBe("exp-n8"); @@ -340,7 +501,7 @@ describe("resolveArmLaunch", () => { expect(launch.timeoutMultiplier).toBe(2); expect(launch.model).toBe("google/gemini-3.5-flash"); expect(launch.role).toBe("variant"); - expect(launch.downshift?.into).toBe("google/gemini-3.5-flash"); + expect(launch.prewalk?.into).toBe("google/gemini-3.5-flash"); }); it("prefers the sibling with a recorded include list over newer include-less siblings", () => { diff --git a/packages/metaharness/src/runner.test.ts b/packages/metaharness/src/runner.test.ts new file mode 100644 index 000000000..4194a8da2 --- /dev/null +++ b/packages/metaharness/src/runner.test.ts @@ -0,0 +1,290 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { + buildHarborEnv, + buildResumeArgs, + collectForwardEnv, + parseArgs, + readTrials, + resolveResumeConfig, +} from "./runner"; + +describe("generic agent-arg / env passthrough", () => { + it("forwards repeated --agent-arg as a JSON array the in-container agent can parse", () => { + const cfg = parseArgs([ + "--model", + "anthropic/claude-opus-4-8", + "--agent-arg", + "--prewalk", + "--agent-arg", + "--prewalk-into", + "--agent-arg", + "google/gemini-3.5-flash", + ]); + expect(cfg.agentArgs).toEqual(["--prewalk", "--prewalk-into", "google/gemini-3.5-flash"]); + + const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); + expect(JSON.parse(env.OMP_BENCH_AGENT_ARGS ?? "[]")).toEqual(cfg.agentArgs); + }); + + it("omits OMP_BENCH_AGENT_ARGS when no --agent-arg was passed", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); + expect(env.OMP_BENCH_AGENT_ARGS).toBeUndefined(); + }); + + it("explicit --providers is authoritative; the default derives from the model", () => { + // Explicit list: exactly what was asked for — the escape hatch that lets + // the model's own provider authenticate directly (forwarded env key) + // while only e.g. oauth-only providers route through the gateway. + const explicit = parseArgs(["--model", "anthropic/claude-opus-4-8", "--providers", "google"]); + const envExplicit = buildHarborEnv(explicit, "/tmp/models.yml", null, "test"); + expect(new Set(envExplicit.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["google"])); + // No flag: the model's provider is gateway-routed by default. + const derived = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + const envDerived = buildHarborEnv(derived, "/tmp/models.yml", null, "test"); + expect(new Set(envDerived.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["anthropic"])); + }); + + it("collects explicit --env pairs, with an explicit value winning over a bare host-forwarded key", () => { + const cfg = parseArgs([ + "--model", + "anthropic/claude-opus-4-8", + "--env", + "SOME_FLAG=1", + "--env", + "OTHER=two words", + ]); + const forwarded = collectForwardEnv(cfg); + expect(forwarded.SOME_FLAG).toBe("1"); + expect(forwarded.OTHER).toBe("two words"); + }); +}); + +describe("install modes", () => { + it("defaults to source mode and publishes the mount contract to the agent", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + expect(cfg.install).toBe("source"); + const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test", { + arch: "arm64", + depsDir: "/tmp/deps", + nodeModules: ["node_modules"], + }); + expect(env.OMP_BENCH_INSTALL).toBe("source"); + expect(env.OMP_BENCH_SOURCE_DIR).toBe("/opt/omp/src"); + expect(env.OMP_BENCH_SOURCE_BUN).toBe("/opt/omp/bin/bun"); + expect(env.OMP_BENCH_SOURCE_ARCH).toBe("arm64"); + }); + + it("omits source mount env when no mount was prepared (binary/local runs)", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--install", "local"]); + const env = buildHarborEnv(cfg, "/tmp/models.yml", "/tmp/omp.tgz", "test"); + expect(env.OMP_BENCH_INSTALL).toBe("local"); + expect(env.OMP_BENCH_SOURCE_DIR).toBeUndefined(); + expect(env.OMP_BENCH_SOURCE_ARCH).toBeUndefined(); + }); + + it("--tarball implies a local (tarball) install", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--tarball", "/tmp/omp.tgz"]); + expect(cfg.install).toBe("local"); + expect(cfg.build).toBe(false); + }); +}); + +describe("parseArgs validation", () => { + it("rejects an unknown flag", () => { + expect(() => parseArgs(["--model", "anthropic/claude-opus-4-8", "--not-a-real-flag"])).toThrow(/unknown flag/); + }); + + it("defaults to a generic, dataset-agnostic jobs directory", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + expect(cfg.jobsDir.endsWith("/runs/harbor")).toBe(true); + }); +}); + +describe("environment backends", () => { + it("defaults to docker with the host.docker.internal gateway", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + expect(cfg.envType).toBe("docker"); + expect(cfg.gatewayUrl).toBe("http://host.docker.internal:4000"); + }); + + it("apple-container swaps the default gateway host to the vmnet bridge address", () => { + const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "apple-container"]); + expect(cfg.envType).toBe("apple-container"); + expect(cfg.gatewayUrl).toBe("http://192.168.64.1:4000"); + }); + + it("an explicit --gateway-url wins over the apple-container default, regardless of flag order", () => { + const cfg = parseArgs([ + "--model", + "anthropic/claude-opus-4-8", + "--gateway-url", + "http://10.0.0.5:9999", + "--environment", + "apple-container", + ]); + expect(cfg.gatewayUrl).toBe("http://10.0.0.5:9999"); + }); + + it("rejects --host-network with apple-container (compose overlay is docker-only)", () => { + expect(() => + parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "apple-container", "--host-network"]), + ).toThrow(/docker-only/); + }); + + it("rejects an invalid --environment value", () => { + expect(() => parseArgs(["--model", "anthropic/claude-opus-4-8", "--environment", "podman"])).toThrow( + /--environment must be/, + ); + }); +}); + +describe("live-trial cost probe", () => { + const usageEvent = (cost: number, input: number, output: number): string => + `${JSON.stringify({ + type: "message_end", + message: { + role: "assistant", + usage: { input, output, cacheRead: 0, cost: { total: cost } }, + }, + })}\n`; + + it("accumulates usage incrementally across appended transcript writes", () => { + const jobDir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-runner-test-")); + try { + const agentDir = path.join(jobDir, "task__abc", "agent"); + fs.mkdirSync(agentDir, { recursive: true }); + const log = path.join(agentDir, "omp.txt"); + + // First flush: one complete event plus a partial line mid-write. + fs.writeFileSync(log, `${usageEvent(0.5, 100, 10)}{"type":"mess`); + let [trial] = readTrials(jobDir); + expect(trial.status).toBe("running"); + expect(trial.costUsd).toBeCloseTo(0.5); + expect(trial.tokIn).toBe(100); + + // Second flush completes the partial line and appends another event. + // Only appended bytes are parsed: the first event must count once. + fs.appendFileSync(log, `age_end"}\n${usageEvent(0.25, 40, 4)}`); + [trial] = readTrials(jobDir); + expect(trial.costUsd).toBeCloseTo(0.75); + expect(trial.tokIn).toBe(140); + expect(trial.tokOut).toBe(14); + } finally { + fs.rmSync(jobDir, { recursive: true, force: true }); + } + }); +}); +describe("resume", () => { + const mkJob = (opts: { + envType?: string; + managerConfig?: Record; + runnerConfig?: Record; + }): { jobsDir: string; jobName: string } => { + const jobsDir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-resume-test-")); + const jobName = "job-x"; + const jobDir = path.join(jobsDir, jobName); + fs.mkdirSync(jobDir, { recursive: true }); + fs.writeFileSync( + path.join(jobDir, "config.json"), + JSON.stringify({ environment: { type: opts.envType ?? "docker" } }), + ); + if (opts.managerConfig) { + fs.writeFileSync( + path.join(jobDir, "manager.json"), + JSON.stringify({ + benchmark: "harbor", + jobName, + dataset: "swe-bench/swe-bench-verified", + config: opts.managerConfig, + }), + ); + } + if (opts.runnerConfig) { + const benchDir = path.join(jobsDir, "_bench", jobName); + fs.mkdirSync(benchDir, { recursive: true }); + fs.writeFileSync(path.join(benchDir, "runner-config.json"), JSON.stringify(opts.runnerConfig)); + } + return { jobsDir, jobName }; + }; + + it("recovers the full launch config from manager.json (API-launched runs)", () => { + const { jobsDir, jobName } = mkJob({ + managerConfig: { + benchmark: "harbor", + model: "openai/gpt-5.6-sol", + include: ["swe-bench/django__django-13837"], + timeoutMultiplier: 2, + extraArgs: ["--providers", "openai-codex", "--agent-arg", "--downshift", "--env", "FOO=bar"], + }, + }); + try { + const cfg = resolveResumeConfig(parseArgs(["--resume", jobName, "--jobs-dir", jobsDir])); + expect(cfg.jobName).toBe(jobName); + expect(cfg.jobsDir).toBe(jobsDir); + expect(cfg.models).toEqual(["openai/gpt-5.6-sol"]); + expect(cfg.dataset).toBe("swe-bench/swe-bench-verified"); + expect(cfg.timeoutMultiplier).toBe(2); + expect(cfg.providers).toEqual(["openai-codex"]); + expect(cfg.agentArgs).toEqual(["--downshift"]); + expect(cfg.env).toEqual({ FOO: "bar" }); + } finally { + fs.rmSync(jobsDir, { recursive: true, force: true }); + } + }); + + it("prefers the runner-config.json snapshot and forces the recorded container backend", () => { + const { jobsDir, jobName } = mkJob({ + envType: "apple-container", + managerConfig: { model: "wrong/model" }, + runnerConfig: { models: ["anthropic/claude-opus-4-8"], envType: "docker" }, + }); + try { + const cfg = resolveResumeConfig( + parseArgs(["--resume", jobName, "--jobs-dir", jobsDir, "--filter-error-type", "RewardFileNotFoundError"]), + ); + expect(cfg.models).toEqual(["anthropic/claude-opus-4-8"]); + // config.json's recorded backend wins, incl. the gateway host swap. + expect(cfg.envType).toBe("apple-container"); + expect(cfg.gatewayUrl).toContain("192.168.64.1"); + // resume-invocation knobs come from the CLI, not the snapshot + expect(cfg.filterErrorTypes).toEqual(["RewardFileNotFoundError"]); + } finally { + fs.rmSync(jobsDir, { recursive: true, force: true }); + } + }); + + it("rejects a job dir without a recorded launch config or without harbor's config.json", () => { + const { jobsDir, jobName } = mkJob({}); + try { + expect(() => resolveResumeConfig(parseArgs(["--resume", jobName, "--jobs-dir", jobsDir]))).toThrow( + /no recorded launch config/, + ); + expect(() => resolveResumeConfig(parseArgs(["--resume", "ghost", "--jobs-dir", jobsDir]))).toThrow( + /no harbor config.json/, + ); + } finally { + fs.rmSync(jobsDir, { recursive: true, force: true }); + } + }); + + it("re-adds harbor's CancelledError default when explicit -f filters would replace it", () => { + const withFilters = parseArgs(["--resume", "j", "--filter-error-type", "RewardFileNotFoundError"]); + expect(buildResumeArgs(withFilters, "/jobs/j")).toEqual([ + "job", + "resume", + "-p", + "/jobs/j", + "-f", + "CancelledError", + "-f", + "RewardFileNotFoundError", + ]); + // No explicit filters → no -f flags: harbor's own default applies. + const bare = parseArgs(["--resume", "j"]); + expect(buildResumeArgs(bare, "/jobs/j")).toEqual(["job", "resume", "-p", "/jobs/j"]); + }); +}); diff --git a/packages/harbor-manager/src/runner.ts b/packages/metaharness/src/runner.ts similarity index 83% rename from packages/harbor-manager/src/runner.ts rename to packages/metaharness/src/runner.ts index 8b7d2b362..1bb8ac385 100755 --- a/packages/harbor-manager/src/runner.ts +++ b/packages/metaharness/src/runner.ts @@ -14,11 +14,12 @@ import * as path from "node:path"; * process renders a live dashboard (progress / success% / spend / tokens / ETA) * by polling each trial's `result.json`. On completion it writes a markdown report. * - * harbor-manager harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 - * harbor-manager harbor --agent oracle --tasks 2 # cheap pipeline smoke - * harbor-manager harbor --help + * metaharness harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 + * metaharness harbor --agent oracle --tasks 2 # cheap pipeline smoke + * metaharness harbor --help */ import type { Server } from "bun"; +import { harborRunnerArgs, type LaunchRequest } from "./launch-args"; // ────────────────────────────────────────────────────────────────────── config @@ -76,6 +77,10 @@ export interface Config { cleanup: boolean; cleanupForce: boolean; hostNetwork: boolean; + /** Job name (or job dir path) to resume via `harbor job resume` instead of starting a new run. */ + resume: string | null; + /** With resume: evict+re-run completed trials that errored with these exception types. */ + filterErrorTypes: string[]; /** Harbor environment backend running the task containers. */ envType: "docker" | "apple-container"; passthrough: string[]; @@ -115,15 +120,17 @@ function defaultConfig(): Config { cleanup: false, cleanupForce: false, hostNetwork: false, + resume: null, + filterErrorTypes: [], envType: "docker", passthrough: [], env: {}, }; } -const HELP = `harbor-manager runner (local omp) +const HELP = `metaharness runner (local omp) -Usage: harbor-manager harbor [options] [-- ] +Usage: metaharness harbor [options] [-- ] Commands: cleanup Force-remove ALL leftover Harbor containers + networks, then exit @@ -166,6 +173,11 @@ Environment: Output / control: -o, --jobs-dir Default /runs/harbor --job-name Default - + --resume Resume that job dir: the original launch flags are recovered + automatically (runner-config.json / manager.json), completed + trials are kept and paid for once, the rest re-run + --filter-error-type With --resume: also re-run completed trials whose exception + type is (repeatable; CancelledError is always evicted) --dry-run Print the harbor command + models.yml and exit --cleanup Clean up stale and exited Harbor Docker resources safely before starting (docker only) --cleanup-force Force-stop and remove ALL previous Harbor Docker containers and networks (docker only) @@ -296,6 +308,12 @@ export function parseArgs(argv: string[]): Config { case "--job-name": cfg.jobName = take(arg); break; + case "--resume": + cfg.resume = take(arg); + break; + case "--filter-error-type": + cfg.filterErrorTypes.push(take(arg)); + break; case "--timeout-multiplier": cfg.timeoutMultiplier = Number(take(arg)); break; @@ -353,6 +371,70 @@ export function parseArgs(argv: string[]): Config { return cfg; } +// ─────────────────────────────────────────────────────────────────── resume + +/** manager.json launch record written by RunStore.registerLaunch. */ +interface ManagerRecord { + benchmark?: string; + dataset?: string; + config?: LaunchRequest; +} + +/** + * Recover the original launch Config for `--resume ` — nothing needs + * re-specifying. Prefers the exact Config snapshot recorded at launch + * (`_bench//runner-config.json`), falling back to rebuilding runner argv + * from the manager.json launch record of API-launched runs. The job dir's own + * harbor config.json decides the container backend: harbor rejects a resume + * whose reconstructed config differs from the recorded one. + */ +export function resolveResumeConfig(cli: Config): Config { + const spec = cli.resume as string; + const jobsDir = spec.includes(path.sep) ? path.dirname(path.resolve(spec)) : cli.jobsDir; + const jobName = path.basename(spec); + const jobDir = path.join(jobsDir, jobName); + const jobConfig = readJson(path.join(jobDir, "config.json")) as { environment?: { type?: string } } | null; + if (!jobConfig) throw new Error(`--resume: ${jobDir} has no harbor config.json (not a harbor job dir)`); + + let cfg: Config | null = null; + const saved = readJson(path.join(jobsDir, "_bench", jobName, "runner-config.json")); + if (saved && typeof saved === "object") { + cfg = { ...defaultConfig(), ...(saved as Partial) }; + } else { + const manager = readJson(path.join(jobDir, "manager.json")) as ManagerRecord | null; + if (manager?.config) { + if (manager.benchmark && manager.benchmark !== "harbor") { + throw new Error(`--resume supports only harbor runs (${jobName} is ${manager.benchmark})`); + } + const dataset = manager.config.dataset ?? manager.dataset ?? "terminal-bench@2.0"; + cfg = parseArgs(harborRunnerArgs(manager.config, { jobsDir, jobName, dataset })); + } + } + if (!cfg) { + throw new Error( + `--resume: no recorded launch config for ${jobName} ` + + `(missing both _bench/${jobName}/runner-config.json and ${jobName}/manager.json)`, + ); + } + cfg.jobsDir = jobsDir; + cfg.jobName = jobName; + cfg.resume = spec; + // The recorded backend wins over any reconstruction-time preference + // (e.g. apple-container auto-detection added after the original run). + const recorded = jobConfig.environment?.type; + if ((recorded === "docker" || recorded === "apple-container") && cfg.envType !== recorded) { + if (recorded === "apple-container" && cfg.gatewayUrl === DOCKER_GATEWAY_URL) cfg.gatewayUrl = VMNET_GATEWAY_URL; + else if (recorded === "docker" && cfg.gatewayUrl === VMNET_GATEWAY_URL) cfg.gatewayUrl = DOCKER_GATEWAY_URL; + cfg.envType = recorded; + } + // Knobs owned by the resume invocation, not the original launch. + cfg.filterErrorTypes = cli.filterErrorTypes; + cfg.passthrough = cli.passthrough; + cfg.dryRun = cli.dryRun; + cfg.cleanup = cli.cleanup; + cfg.cleanupForce = cli.cleanupForce; + return cfg; +} // ──────────────────────────────────────────────────────────────────── helpers const isTTY = Boolean(process.stdout.isTTY); @@ -444,6 +526,114 @@ function readJson(file: string): unknown { } } +/** Running usage totals for one live trial's transcript, plus the parse cursor. */ +interface CostProbe { + /** Bytes of the transcript already consumed. */ + offset: number; + /** Trailing partial line carried to the next read (bytes, so multi-byte chars survive chunking). */ + remainder: Buffer; + /** True while discarding an oversized line (resync at the next newline). */ + discarding: boolean; + costUsd: number; + tokIn: number; + tokOut: number; + tokCache: number; +} + +/** Incremental parse state per live transcript path. Entries are dropped once the trial finishes. */ +const costProbes = new Map(); + +/** First sight of an already-huge transcript: parse only its tail (undercounts cost, never OOMs). */ +const COST_PROBE_FIRST_SCAN_BYTES = 16 * 1024 * 1024; +/** A single line longer than this is bloat/corruption, never a usage event: skip it. */ +const COST_PROBE_MAX_LINE_BYTES = 4 * 1024 * 1024; +const COST_PROBE_CHUNK_BYTES = 1024 * 1024; + +/** Accumulate assistant `message_end` usage from one complete transcript line. */ +function probeLine(line: string, probe: CostProbe): void { + const trimmed = line.trim(); + if (!trimmed) return; + try { + const event = JSON.parse(trimmed); + if (event?.type !== "message_end") return; + const message = event.message; + if (!message || typeof message !== "object" || message.role !== "assistant") return; + const usage = message.usage; + if (!usage || typeof usage !== "object") return; + probe.tokIn += num(usage.input) + num(usage.cacheRead); + probe.tokOut += num(usage.output); + probe.tokCache += num(usage.cacheRead); + const cost = usage.cost; + if (cost && typeof cost === "object") probe.costUsd += num(cost.total); + } catch { + /* Ignore malformed lines from incomplete writes */ + } +} + +/** + * Realtime usage for a still-running trial, read incrementally from its + * `agent/omp.txt` JSONL. Only bytes appended since the previous call are read + * and parsed — both this runner's render loop and the manager's 2s sync tick + * call this for every live trial, and a full-file reread used to block the + * event loop for seconds (and OOM outright on runaway multi-GB transcripts). + */ +function probeTrialCost(ompLogPath: string): CostProbe | null { + let size: number; + try { + size = fs.statSync(ompLogPath).size; + } catch { + return costProbes.get(ompLogPath) ?? null; + } + let probe = costProbes.get(ompLogPath); + if (!probe || size < probe.offset) { + // New (or truncated/rotated) transcript. Skip a pre-existing giant head. + probe = { + offset: Math.max(0, size - COST_PROBE_FIRST_SCAN_BYTES), + remainder: Buffer.alloc(0), + discarding: size > COST_PROBE_FIRST_SCAN_BYTES, // resync to the next full line + costUsd: 0, + tokIn: 0, + tokOut: 0, + tokCache: 0, + }; + costProbes.set(ompLogPath, probe); + } + if (size === probe.offset) return probe; + let fd: number; + try { + fd = fs.openSync(ompLogPath, "r"); + } catch { + return probe; + } + try { + const chunk = Buffer.allocUnsafe(COST_PROBE_CHUNK_BYTES); + for (;;) { + const read = fs.readSync(fd, chunk, 0, chunk.length, probe.offset); + if (read <= 0) break; + probe.offset += read; + const data = Buffer.concat([probe.remainder, chunk.subarray(0, read)]); + let start = 0; + for (;;) { + const nl = data.indexOf(0x0a, start); + if (nl === -1) break; + if (probe.discarding) probe.discarding = false; + else probeLine(data.subarray(start, nl).toString("utf8"), probe); + start = nl + 1; + } + probe.remainder = data.subarray(start); + if (probe.remainder.length > COST_PROBE_MAX_LINE_BYTES) { + probe.remainder = Buffer.alloc(0); + probe.discarding = true; + } + } + } catch { + /* keep whatever was accumulated; retry next tick */ + } finally { + fs.closeSync(fd); + } + return probe; +} + /** Parse one trial directory into a Trial, or null if it isn't a trial dir yet. */ function parseTrial(dir: string, name: string): Trial | null { const resultPath = path.join(dir, "result.json"); @@ -456,43 +646,12 @@ function parseTrial(dir: string, name: string): Trial | null { /* ignore */ } - // Try to parse realtime cost from the live agent omp.txt log if it exists - let costUsd = 0; - let tokIn = 0; - let tokOut = 0; - let tokCache = 0; - const ompLogPath = path.join(dir, "agent", "omp.txt"); - if (fs.existsSync(ompLogPath)) { - try { - const content = fs.readFileSync(ompLogPath, "utf8"); - for (const line of content.split("\n")) { - const trimmed = line.trim(); - if (!trimmed) continue; - try { - const event = JSON.parse(trimmed); - if (event && event.type === "message_end") { - const message = event.message; - if (message && typeof message === "object" && message.role === "assistant") { - const usage = message.usage; - if (usage && typeof usage === "object") { - tokIn += num(usage.input) + num(usage.cacheRead); - tokOut += num(usage.output); - tokCache += num(usage.cacheRead); - const cost = usage.cost; - if (cost && typeof cost === "object") { - costUsd += num(cost.total); - } - } - } - } - } catch { - /* Ignore malformed lines from incomplete writes */ - } - } - } catch { - /* ignore */ - } - } + // Realtime cost from the live agent omp.txt log, parsed incrementally. + const probe = probeTrialCost(path.join(dir, "agent", "omp.txt")); + const costUsd = probe?.costUsd ?? 0; + const tokIn = probe?.tokIn ?? 0; + const tokOut = probe?.tokOut ?? 0; + const tokCache = probe?.tokCache ?? 0; return { name, @@ -506,6 +665,8 @@ function parseTrial(dir: string, name: string): Trial | null { detail: "", }; } + // Trial finished: usage now comes from result.json; drop the live-parse state. + costProbes.delete(path.join(dir, "agent", "omp.txt")); const raw = readJson(resultPath); if (!raw || typeof raw !== "object") return null; const r = raw as Record; @@ -1059,7 +1220,13 @@ function buildMountsJson(source: SourceMount | null): string | null { } function deriveProviders(cfg: Config): string[] { - const set = new Set(cfg.providers); + // Explicit --providers is authoritative: it's the escape hatch for routing + // only SOME providers through the gateway (e.g. oauth-only openai-codex) + // while the model's own provider authenticates directly via a forwarded + // env key. The model-provider + anthropic/openai-codex additions are the + // DEFAULT for when the flag is absent. + if (cfg.providers.length > 0) return [...new Set(cfg.providers)]; + const set = new Set(); for (const m of cfg.models) { const slash = m.indexOf("/"); if (slash > 0) set.add(m.slice(0, slash)); @@ -1073,7 +1240,7 @@ function deriveProviders(cfg: Config): string[] { function writeModelsYaml(benchDir: string, cfg: Config): string { const providers = deriveProviders(cfg); - const lines = ["# Generated by harbor-manager — auth via host pm2 gateway.", "providers:"]; + const lines = ["# Generated by metaharness — auth via host pm2 gateway.", "providers:"]; for (const p of providers) { lines.push(` ${p}:`); lines.push(` baseUrl: ${cfg.gatewayUrl}`); @@ -1170,6 +1337,20 @@ function buildHarborArgs( a.push(...cfg.passthrough); return a; } +/** + * `harbor job resume` argv for an existing job dir: trial dirs with a + * result.json are kept (their spend is reused), the rest re-run. Explicit + * `-f` values REPLACE harbor's CancelledError default, so it is always + * re-added alongside the caller's filters. + */ +export function buildResumeArgs(cfg: Config, jobDir: string): string[] { + const a: string[] = ["job", "resume", "-p", jobDir]; + if (cfg.filterErrorTypes.length > 0) { + for (const t of new Set(["CancelledError", ...cfg.filterErrorTypes])) a.push("-f", t); + } + a.push(...cfg.passthrough); + return a; +} const FORWARD_ENV_DENYLIST = new Set([ "PI_CODING_AGENT_DIR", @@ -1376,6 +1557,11 @@ async function runBenchmark(cfg: Config): Promise { const jobDir = path.join(cfg.jobsDir, jobName); const benchDir = path.join(cfg.jobsDir, "_bench", jobName); fs.mkdirSync(benchDir, { recursive: true }); + if (!cfg.resume && !cfg.dryRun) { + // Snapshot the resolved launch config so a later `--resume ` can + // rebuild the exact same invocation without re-specifying flags. + fs.writeFileSync(path.join(benchDir, "runner-config.json"), JSON.stringify({ ...cfg, jobName }, null, "\t")); + } const version = readPkgVersion(); @@ -1413,7 +1599,9 @@ async function runBenchmark(cfg: Config): Promise { const composeOverlayPath = cfg.envType === "docker" ? writeComposeOverlay(benchDir, cfg, source) : null; const mountsJson = cfg.envType === "docker" ? null : buildMountsJson(source); - const harborArgs = buildHarborArgs(cfg, jobName, modelsYaml, tarball, composeOverlayPath, mountsJson); + const harborArgs = cfg.resume + ? buildResumeArgs(cfg, jobDir) + : buildHarborArgs(cfg, jobName, modelsYaml, tarball, composeOverlayPath, mountsJson); const harborEnv = buildHarborEnv(cfg, modelsYaml, tarball, version, source); const logPath = path.join(benchDir, "harbor.log"); if (cfg.dryRun) { @@ -1455,7 +1643,9 @@ async function runBenchmark(cfg: Config): Promise { stdin: "ignore", }); - const expected = Math.max(1, cfg.tasks * cfg.attempts * cfg.models.length); + const expected = cfg.resume + ? (readJobResult(jobDir)?.nTotal ?? Math.max(1, cfg.tasks * cfg.attempts * cfg.models.length)) + : Math.max(1, cfg.tasks * cfg.attempts * cfg.models.length); const st: RenderState = { cfg, jobDir, logPath, startMs: Date.now(), expected, tick: 0 }; if (isTTY) process.stdout.write(`${ESC}?1049h${ESC}?25l`); // alt screen, hide cursor @@ -1525,7 +1715,8 @@ async function main(): Promise { runDockerCleanup(true); return; } - const cfg = parseArgs(argv); + let cfg = parseArgs(argv); + if (cfg.resume) cfg = resolveResumeConfig(cfg); const exitCode = (await runBenchmark(cfg)).exitCode; process.exit(exitCode); } diff --git a/packages/harbor-manager/src/server.ts b/packages/metaharness/src/server.ts similarity index 58% rename from packages/harbor-manager/src/server.ts rename to packages/metaharness/src/server.ts index 7ccca4368..17d31b513 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/metaharness/src/server.ts @@ -1,16 +1,23 @@ #!/usr/bin/env bun /** - * harbor-manager server: REST + SSE API over the run store, static web + * metaharness server: REST + SSE API over the run store, static web * dashboard, and a launcher that spawns the CLI runner as a managed child. * * bun src/server.ts [--port 4700] [--jobs-dir ] * * API: - * GET /api/experiments → experiment summaries across all benchmarks - * GET /api/runs → RunRow[] + * GET /api/experiments[?q=] → experiment summaries across all benchmarks + * POST /api/experiments → register an experiment (id + goal) before its first arm + * GET /api/experiments/:id → experiment detail (arms, task matrix) + * PUT /api/experiments/:id → update goal + per-run role/note/label + * DELETE /api/experiments/:id → delete all arms (rows + job dirs) and the goal row + * POST /api/experiments/:id/arms → launch a comparable arm + * GET /api/runs[?experiment=&status=&benchmark=] → RunRow[] * POST /api/runs → launch any benchmark * GET /api/runs/:name → { run, traces } - * DELETE /api/runs/:name → cancel a managed run + * POST /api/runs/:name/cancel → cancel a managed run + * POST /api/runs/:name/resume → resume an incomplete harbor run + * DELETE /api/runs/:name → delete a finished run (row + job dir) * GET /api/runs/:name/traces/:trace → normalized trace * GET /api/events → SSE: run-list snapshots on change */ @@ -19,7 +26,8 @@ import * as path from "node:path"; import type { Server, Subprocess } from "bun"; import { BENCHMARK_DEFINITIONS } from "./benchmarks"; import { buildExperiments, experimentDetail, experimentOf } from "./experiments"; -import { type BenchmarkKind, type RunRole, type RunRow, RunStore } from "./store"; +import { harborRunnerArgs, type LaunchRequest } from "./launch-args"; +import { type LaunchRecord, type RunRole, type RunRow, RunStore } from "./store"; /** PUT /api/experiments/:id body — goal and per-run role/note/label metadata. */ export interface ExperimentMetaUpdate { @@ -27,50 +35,27 @@ export interface ExperimentMetaUpdate { runs?: Record; } -const INDEX_HTML_PATH = new URL("./web/index.html", import.meta.url).pathname; +/** POST /api/experiments body — pre-registers an experiment id with a goal. */ +export interface CreateExperimentRequest { + /** Dash-free token; runs group into it as `-` job names. */ + id: string; + goal?: string; +} + +import indexHtml from "./web/index.html"; const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); const PKG_DIR = path.resolve(import.meta.dir, ".."); const DEFAULT_JOBS_DIR = path.join(REPO_ROOT, "runs", "harbor"); -/** POST /api/runs body. Mirrors the runner CLI surface we actually use. */ -export interface LaunchRequest { - /** Benchmark adapter to execute. */ - benchmark?: BenchmarkKind; - model: string; - dataset?: string; - /** Task count for a dataset sample, or omit when `include` is given. */ - tasks?: number; - /** Explicit task names (passed as repeated --include). */ - include?: string[]; - concurrency?: number; - /** SnapCompact conditions; ignored by other benchmarks. */ - conditions?: string[]; - timeoutMultiplier?: number; - attempts?: number; - agent?: string; - jobName?: string; - webSearch?: boolean; - /** Downshift to a fast/cheap model at the first edit/write once the todo list exists; `into` overrides the default "smol" target. */ - downshift?: { into?: string }; - /** Role of this run inside its experiment (baseline vs treatment). */ - role?: RunRole; - /** One-line description of what this arm tests. */ - note?: string; - /** Experiment goal; upserted for the run's experiment (job-name prefix). */ - goal?: string; - /** Use prebuilt dist/omp-linux-* binaries instead of the default source mount. */ - prebuiltBinaries?: boolean; - /** Extra raw runner args, appended verbatim. */ - extraArgs?: string[]; -} +export type { LaunchRequest } from "./launch-args"; /** POST /api/experiments/:id/arms body — a new comparable arm; sample+config inherited. */ export interface AddArmRequest { /** Arm label; becomes the `-` job name. */ arm: string; model: string; - downshift?: LaunchRequest["downshift"]; + prewalk?: LaunchRequest["prewalk"]; /** Explicit task sample; skips sibling inheritance when provided. */ include?: string[]; role?: RunRole; @@ -105,12 +90,30 @@ function parseServerArgs(argv: string[]): { port: number; jobsDir: string } { return { port, jobsDir }; } +/** Job names are single path segments; anything else could escape the jobs dir. */ +function assertSafeJobName(jobName: string): void { + if (!jobName || jobName === "." || jobName === ".." || /[/\\]/.test(jobName)) { + throw new Error(`invalid job name: ${jobName}`); + } +} + +/** True when `pid` names a live process (signal-0 probe). */ +function pidAlive(pid: number | null): boolean { + if (pid == null) return false; + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +} + /** * Resolve the launch request for a new arm added to an existing experiment. * Inherits the experiment's benchmark, dataset, and — crucially — the exact * task sample from a sibling arm (its recorded `include`, else its observed * trial tasks) so the arm is directly comparable. Only per-arm knobs (model, - * downshift, role, note, extra args) come from `req`. Throws if the experiment has + * prewalk, role, note, extra args) come from `req`. Throws if the experiment has * no runs to inherit from or the arm name is taken. */ export function resolveArmLaunch(store: RunStore, experimentId: string, req: AddArmRequest): LaunchRequest { @@ -174,9 +177,10 @@ export function resolveArmLaunch(store: RunStore, experimentId: string, req: Add prebuiltBinaries: cfg.prebuiltBinaries === true || undefined, conditions: conditions.length > 0 ? conditions : undefined, jobName, - downshift: req.downshift, + prewalk: req.prewalk, role: req.role, note: req.note, + environment: cfg.environment === "docker" || cfg.environment === "apple-container" ? cfg.environment : undefined, extraArgs: req.extraArgs, }; } @@ -188,7 +192,7 @@ export class ManagerServer { #lastSnapshot = ""; #syncTimer: Timer | undefined; #server: Server | null = null; - #appBundleCode: string | null = null; + #stopped = false; readonly jobsDir: string; constructor(jobsDir: string, dbPath?: string) { @@ -207,12 +211,20 @@ export class ManagerServer { this.#server = Bun.serve({ port, idleTimeout: 0, + // Bun bundles the dashboard (React + TSX) from the HTML import and + // serves it on the same port as the API — one process, no Vite. + routes: { "/": indexHtml }, + // Only `hmr`: the `console: true` mirror adds another dev-client + // stream with the same teardown hazard (see isDevStreamTeardown) + // for little value. + development: process.env.NODE_ENV !== "production" && { hmr: true }, fetch: request => this.#route(request), }); return this.#server; } async stop(): Promise { + this.#stopped = true; clearInterval(this.#syncTimer); for (const client of this.#sse) { client.state = SseState.Closed; @@ -234,22 +246,6 @@ export class ManagerServer { } } - /** Bundle the React dashboard once per process; served at /app.tsx (matches the Vite dev entry). */ - async #appBundle(): Promise { - if (this.#appBundleCode !== null) return this.#appBundleCode; - const result = await Bun.build({ - entrypoints: [path.join(import.meta.dir, "web", "app.tsx")], - target: "browser", - minify: true, - define: { "process.env.NODE_ENV": '"production"' }, - }); - if (!result.success) { - throw new Error(`dashboard bundle failed:\n${result.logs.map(l => l.message).join("\n")}`); - } - this.#appBundleCode = await result.outputs[0].text(); - return this.#appBundleCode; - } - #broadcast(frame: string): void { const bytes = new TextEncoder().encode(frame); for (const client of this.#sse) { @@ -267,20 +263,22 @@ export class ManagerServer { const url = new URL(request.url); const p = url.pathname; try { - if (p === "/" || p === "/index.html") { - return new Response(Bun.file(INDEX_HTML_PATH)); - } - if (p === "/app.tsx") { - return new Response(await this.#appBundle(), { - headers: { "content-type": "text/javascript; charset=utf-8" }, - }); - } if (p === "/api/events") return this.#sseResponse(); if (p === "/api/benchmarks" && request.method === "GET") { return Response.json(BENCHMARK_DEFINITIONS); } if (p === "/api/experiments" && request.method === "GET") { - return Response.json(buildExperiments(this.#store)); + const q = url.searchParams.get("q")?.toLowerCase() ?? ""; + const experiments = buildExperiments(this.#store); + return Response.json( + q + ? experiments.filter(e => e.id.toLowerCase().includes(q) || e.goal.toLowerCase().includes(q)) + : experiments, + ); + } + if (p === "/api/experiments" && request.method === "POST") { + const body = (await request.json()) as CreateExperimentRequest; + return Response.json(this.createExperiment(body), { status: 201 }); } const expMatch = p.match(/^\/api\/experiments\/([^/]+)$/); if (expMatch) { @@ -289,6 +287,11 @@ export class ManagerServer { const body = (await request.json()) as ExperimentMetaUpdate; return Response.json(this.updateExperimentMeta(id, body)); } + if (request.method === "DELETE") { + const result = this.deleteExperiment(id); + if (!result) return Response.json({ error: "experiment not found" }, { status: 404 }); + return Response.json(result); + } const detail = experimentDetail(this.#store, id); if (!detail) return Response.json({ error: "experiment not found" }, { status: 404 }); return Response.json(detail); @@ -300,16 +303,36 @@ export class ManagerServer { return Response.json(this.addArm(id, body), { status: 201 }); } if (p === "/api/runs" && request.method === "GET") { - return Response.json(this.#store.listRuns()); + const experiment = url.searchParams.get("experiment"); + const status = url.searchParams.get("status"); + const benchmark = url.searchParams.get("benchmark"); + let runs = this.#store.listRuns(); + if (experiment) runs = runs.filter(r => experimentOf(r.jobName) === experiment); + if (status) runs = runs.filter(r => r.status === status); + if (benchmark) runs = runs.filter(r => r.benchmark === benchmark); + return Response.json(runs); } if (p === "/api/runs" && request.method === "POST") { const body = (await request.json()) as LaunchRequest; return Response.json(this.launch(body), { status: 201 }); } + const resumeMatch = p.match(/^\/api\/runs\/([^/]+)\/resume$/); + if (resumeMatch && request.method === "POST") { + const jobName = decodeURIComponent(resumeMatch[1]); + const body = (await request.json().catch(() => ({}))) as { filterErrorTypes?: string[] }; + return Response.json(this.resume(jobName, body), { status: 201 }); + } + const cancelMatch = p.match(/^\/api\/runs\/([^/]+)\/cancel$/); + if (cancelMatch && request.method === "POST") { + return Response.json(this.cancel(decodeURIComponent(cancelMatch[1]))); + } const runMatch = p.match(/^\/api\/runs\/([^/]+)$/); if (runMatch) { const jobName = decodeURIComponent(runMatch[1]); - if (request.method === "DELETE") return Response.json(this.cancel(jobName)); + if (request.method === "DELETE") { + if (!this.deleteRun(jobName)) return Response.json({ error: "run not found" }, { status: 404 }); + return Response.json({ jobName, deleted: true }); + } const run = this.#store.syncRun(jobName); if (!run) return Response.json({ error: "run not found" }, { status: 404 }); return Response.json({ run, traces: this.#store.listTraces(jobName) }); @@ -389,43 +412,74 @@ export class ManagerServer { if (request.conditions?.length) argv.push("--conditions", request.conditions.join(",")); } else { cwd = PKG_DIR; - argv = [ - "bun", - "src/runner.ts", - "--model", - request.model, - "-d", - dataset, - "--job-name", - jobName, - "--jobs-dir", - this.jobsDir, - ]; - if (request.agent) argv.push("--agent", request.agent); - if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks)); - if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency)); - if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts)); - if (request.timeoutMultiplier !== undefined) - argv.push("--timeout-multiplier", String(request.timeoutMultiplier)); - if (request.webSearch) argv.push("--web-search"); - for (const task of request.include ?? []) argv.push("--include", task); - if (request.downshift) { - argv.push("--agent-arg", "--downshift"); - if (request.downshift.into) { - argv.push("--agent-arg", "--downshift-into", "--agent-arg", request.downshift.into); - const provider = request.downshift.into.split("/", 1)[0]; - if (provider && request.downshift.into.includes("/")) argv.push("--providers", provider); - } - } - if (request.prebuiltBinaries) { - for (const name of ["omp-linux-arm64", "omp-linux-x64"]) { - const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name); - if (fs.existsSync(binary)) argv.push("--binary", binary); - } - } + argv = ["bun", "src/runner.ts", ...harborRunnerArgs(request, { jobsDir: this.jobsDir, jobName, dataset })]; } - argv.push(...(request.extraArgs ?? [])); + if (benchmark !== "harbor") argv.push(...(request.extraArgs ?? [])); + const pid = this.#spawnRunner(argv, cwd, { + benchmark, + jobName, + dataset, + agent: request.agent ?? "omp", + models: [request.model], + prewalk: request.prewalk, + config: { ...request }, + role: request.role, + note: request.note, + }); + if (request.goal) this.#store.setExperimentGoal(experimentOf(jobName), request.goal); + return { jobName, pid }; + } + + /** + * Resume a harbor run in place via the runner's `--resume`: completed + * trials (and their spend) are reused, interrupted/pending trials re-run, + * and errored trials are evicted for retry. The runner recovers the + * original launch flags from the run's recorded config, so nothing needs + * re-specifying. `filterErrorTypes` overrides the default retry set + * (every exception type recorded in the job's result.json). + */ + resume(jobName: string, opts: { filterErrorTypes?: string[] } = {}): { jobName: string; pid: number } { + const run = this.#store.getRun(jobName); + if (!run) throw new Error(`run ${jobName} not found`); + if (run.benchmark !== "harbor") + throw new Error(`resume supports only harbor runs (${jobName} is ${run.benchmark})`); + // Trust liveness, not the recorded status: a runner killed while a + // previous server instance owned it leaves a stale `running` row with a + // dead (or null) pid and nobody to fire markExit. + if (this.#runLive(run)) { + throw new Error(`run ${jobName} is already running`); + } + if (run.status === "running") this.#store.markExit(jobName, null, true); + const jobDir = path.join(this.jobsDir, jobName); + if (!fs.existsSync(path.join(jobDir, "config.json"))) { + throw new Error(`${jobName} has no harbor config.json to resume from`); + } + const argv = ["bun", "src/runner.ts", "--resume", jobName, "--jobs-dir", this.jobsDir]; + for (const t of opts.filterErrorTypes ?? erroredExceptionTypes(jobDir)) argv.push("--filter-error-type", t); + let prewalk: LaunchRequest["prewalk"]; + try { + prewalk = run.prewalk ? (JSON.parse(run.prewalk) as { into?: string }) : undefined; + } catch { + prewalk = undefined; + } + const pid = this.#spawnRunner(argv, PKG_DIR, { + benchmark: "harbor", + jobName, + dataset: run.dataset, + agent: run.agent, + models: run.models ? run.models.split(",") : [], + prewalk, + config: run.config, + role: run.role, + note: run.note, + }); + return { jobName, pid }; + } + + /** Spawn a detached runner child, wire its exit back into the store, and register the run. */ + #spawnRunner(argv: string[], cwd: string, record: Omit): number { + const jobName = record.jobName; const logDir = path.join(this.jobsDir, "_manager", "logs"); fs.mkdirSync(logDir, { recursive: true }); const logFile = fs.openSync(path.join(logDir, `${jobName}.log`), "w"); @@ -434,10 +488,19 @@ export class ManagerServer { stdout: logFile, stderr: logFile, env: { ...process.env }, + // Own process group: a manager restart (Ctrl+C / --hot dev cycle) must + // not deliver terminal signals to runners — that killed live runs. + detached: true, }); const child: ManagedChild = { proc, jobName, cancelled: false }; this.#children.set(jobName, child); proc.exited.then(exitCode => { + try { + fs.closeSync(logFile); + } catch {} + // A retired instance (--hot reload) must not touch the closed store; + // the successor's pid sweep reconciles this run from disk instead. + if (this.#stopped) return; this.#store.markExit(jobName, exitCode, child.cancelled); // Final sync AFTER the terminal state: the ticker only revisits // running rows, so the last-2s trial results would otherwise be lost. @@ -445,35 +508,85 @@ export class ManagerServer { this.#children.delete(jobName); this.#tick(); }); - this.#store.registerLaunch({ - benchmark, - jobName, - dataset, - agent: request.agent ?? "omp", - models: [request.model], - downshift: request.downshift, - config: { ...request }, - pid: proc.pid, - role: request.role, - note: request.note, - }); - if (request.goal) this.#store.setExperimentGoal(experimentOf(jobName), request.goal); + this.#store.registerLaunch({ ...record, pid: proc.pid }); this.#tick(); - return { jobName, pid: proc.pid }; + return proc.pid; + } + + /** Liveness check that survives manager restarts: managed child, or a running row with a live pid. */ + #runLive(run: RunRow): boolean { + return this.#children.has(run.jobName) || (run.status === "running" && pidAlive(run.pid)); + } + + /** Register an experiment id (with an optional goal) so it is browsable before its first arm. */ + createExperiment(req: CreateExperimentRequest): { id: string; goal: string } { + const id = req.id?.trim() ?? ""; + // Dashes are structurally impossible: `experimentOf` groups job names by + // the token before the first dash, so a dashed id could never own a run. + if (!/^[A-Za-z0-9_.]+$/.test(id)) { + throw new Error("experiment id must be a non-empty token of [A-Za-z0-9_.] (runs group as `-`)"); + } + const goal = req.goal ?? this.#store.getExperimentMeta(id)?.goal ?? ""; + this.#store.setExperimentGoal(id, goal); + return { id, goal }; } /** Apply goal + per-run role/note metadata; used by the UI and for backfill. */ updateExperimentMeta(id: string, update: ExperimentMetaUpdate): { id: string; updatedRuns: string[] } { if (update.goal !== undefined) this.#store.setExperimentGoal(id, update.goal); const updatedRuns: string[] = []; - for (const [jobName, meta] of Object.entries(update.runs ?? {})) { + for (const jobName in update.runs) { if (experimentOf(jobName) !== id) continue; - if (this.#store.setRunMeta(jobName, meta)) updatedRuns.push(jobName); + if (this.#store.setRunMeta(jobName, update.runs[jobName])) updatedRuns.push(jobName); } this.#tick(); return { id, updatedRuns }; } + /** + * Delete an experiment: every arm's DB row, job dir, and manager log, plus + * the goal row. Refuses while any arm is live (cancel first — deleting a + * job dir under a writing runner would corrupt it). Returns null when the + * id names neither runs nor a registered experiment. + */ + deleteExperiment(id: string): { id: string; deletedRuns: string[] } | null { + const runs = this.#store.listRuns().filter(r => experimentOf(r.jobName) === id); + if (runs.length === 0 && !this.#store.getExperimentMeta(id)) return null; + const live = runs.filter(r => this.#runLive(r)); + if (live.length > 0) { + throw new Error( + `experiment ${id} has running arms (${live.map(r => r.jobName).join(", ")}); cancel them first`, + ); + } + for (const run of runs) this.#destroyRun(run.jobName); + this.#store.deleteExperimentMeta(id); + this.#tick(); + return { id, deletedRuns: runs.map(r => r.jobName) }; + } + + /** + * Permanently delete a run: DB row + trials, job dir, and manager log. + * Disk removal is not optional — discover() would resurrect a surviving + * job dir as a fresh row on the next restart. Refuses while the run is + * live; returns false when the run is unknown. + */ + deleteRun(jobName: string): boolean { + const run = this.#store.getRun(jobName); + if (!run) return false; + if (this.#runLive(run)) throw new Error(`run ${jobName} is running; cancel it first`); + this.#destroyRun(jobName); + this.#tick(); + return true; + } + + /** Remove a run's DB rows and on-disk artifacts (job dir + manager log). */ + #destroyRun(jobName: string): void { + assertSafeJobName(jobName); + this.#store.deleteRun(jobName); + fs.rmSync(path.join(this.jobsDir, jobName), { recursive: true, force: true }); + fs.rmSync(path.join(this.jobsDir, "_manager", "logs", `${jobName}.log`), { force: true }); + } + /** Add a comparable arm to an existing experiment, inheriting its sample + config. */ addArm(experimentId: string, req: AddArmRequest): { jobName: string; pid: number } { return this.launch(resolveArmLaunch(this.#store, experimentId, req)); @@ -540,7 +653,7 @@ export class ManagerServer { if (!file.startsWith(`${path.resolve(jobDir)}${path.sep}`) || !fs.existsSync(file)) { return Response.json({ error: "trace not found" }, { status: 404 }); } - const text = fs.readFileSync(file, "utf8"); + const text = readTextTail(file, TRACE_READ_CAP_BYTES); if (!file.endsWith(".txt")) { if (raw) return new Response(text, { headers: { "content-type": "text/plain; charset=utf-8" } }); return Response.json({ @@ -590,16 +703,83 @@ export class ManagerServer { return Response.json({ jobName, trace: traceName, entries: entries.slice(-n), totalEvents: lines.length }); } } +/** + * Exception types recorded in a job's result.json — the errored trials a + * resume retries by default (reward-0 fails are completed results and stay). + */ +function erroredExceptionTypes(jobDir: string): string[] { + try { + const raw = JSON.parse(fs.readFileSync(path.join(jobDir, "result.json"), "utf8")) as { + stats?: { evals?: Record }> }; + }; + const types = new Set(); + for (const ev of Object.values(raw.stats?.evals ?? {})) { + for (const t of Object.keys(ev.exception_stats ?? {})) types.add(t); + } + return [...types]; + } catch { + return []; + } +} + +/** Trace files can be runaway-huge; the viewer only shows a tail anyway. */ +const TRACE_READ_CAP_BYTES = 32 * 1024 * 1024; + +/** Last `cap` bytes of a file as text, dropping a leading partial line when truncated. */ +function readTextTail(file: string, cap: number): string { + const size = fs.statSync(file).size; + if (size <= cap) return fs.readFileSync(file, "utf8"); + const fd = fs.openSync(file, "r"); + try { + const buf = Buffer.allocUnsafe(cap); + const read = fs.readSync(fd, buf, 0, cap, size - cap); + const text = buf.subarray(0, read).toString("utf8"); + const nl = text.indexOf("\n"); + return nl === -1 ? text : text.slice(nl + 1); + } finally { + fs.closeSync(fd); + } +} + +/** + * Bun's dev server (HMR websocket, browser error reports, console mirror) + * reads client streams that a tab disconnect or `--hot` reload tears down + * mid-read. The resulting `AbortError: ERR_STREAM_RELEASE_LOCK` surfaces as + * an unhandled rejection from Bun internals — fatal by default, which would + * kill the manager and orphan every running benchmark job. + */ +function isDevStreamTeardown(err: unknown): boolean { + return err instanceof Error && (err as Error & { code?: string }).code === "ERR_STREAM_RELEASE_LOCK"; +} if (import.meta.main) { + // `bun --hot` re-evaluates this module in-place: retire the previous + // instance first, or its sync ticker and sqlite connection leak per reload. + const host = globalThis as typeof globalThis & { + __metaharnessServer?: ManagerServer; + __metaharnessHooks?: boolean; + }; + await host.__metaharnessServer?.stop(); const { port, jobsDir } = parseServerArgs(process.argv.slice(2)); const manager = new ManagerServer(jobsDir); + host.__metaharnessServer = manager; const server = manager.start(port); - process.stdout.write(`harbor-manager listening on http://localhost:${server.port} (jobs: ${jobsDir})\n`); - const shutdown = async () => { - await manager.stop(); - process.exit(0); - }; - process.on("SIGINT", shutdown); - process.on("SIGTERM", shutdown); + process.stdout.write(`metaharness listening on http://localhost:${server.port} (jobs: ${jobsDir})\n`); + // Process-wide hooks register once; `--hot` re-evals reuse them via `host`. + if (!host.__metaharnessHooks) { + host.__metaharnessHooks = true; + const shutdown = async () => { + await host.__metaharnessServer?.stop(); + process.exit(0); + }; + process.on("SIGINT", shutdown); + process.on("SIGTERM", shutdown); + process.on("unhandledRejection", err => { + if (isDevStreamTeardown(err)) { + process.stderr.write("ignored dev-server stream teardown (ERR_STREAM_RELEASE_LOCK)\n"); + return; + } + throw err; // preserve fail-fast for real bugs + }); + } } diff --git a/packages/harbor-manager/src/store.ts b/packages/metaharness/src/store.ts similarity index 77% rename from packages/harbor-manager/src/store.ts rename to packages/metaharness/src/store.ts index cf3653375..c06f24fc7 100644 --- a/packages/harbor-manager/src/store.ts +++ b/packages/metaharness/src/store.ts @@ -27,13 +27,13 @@ export interface RunRow { dataset: string; agent: string; models: string; - /** JSON downshift config (`{ into?: string }`); older rows may hold legacy reasoning-slide JSON. */ - downshift: string | null; + /** JSON prewalk config (`{ into?: string }`); older rows may hold legacy reasoning-slide JSON. */ + prewalk: string | null; /** Benchmark-specific launch configuration. */ config: Record; /** Role inside the experiment (baseline vs treatment); "" when unspecified. */ role: RunRole; - /** One-line description of what this arm tests (e.g. "downshift→flash at first edit/write"). */ + /** One-line description of what this arm tests (e.g. "prewalk→flash at first edit/write"). */ note: string; /** Display-name override for the arm; "" falls back to the jobName-derived arm label. */ label: string; @@ -72,13 +72,20 @@ export interface TraceRow { tracePath: string | null; } +/** Row in the `experiments` table: goal metadata keyed by experiment id. */ +export interface ExperimentMeta { + id: string; + goal: string; + updatedAt: number; +} + export interface LaunchRecord { benchmark: BenchmarkKind; jobName: string; dataset: string; agent: string; models: string[]; - downshift?: { into?: string }; + prewalk?: { into?: string }; pid: number; role?: RunRole; note?: string; @@ -92,7 +99,7 @@ CREATE TABLE IF NOT EXISTS runs ( dataset TEXT NOT NULL DEFAULT '', agent TEXT NOT NULL DEFAULT 'omp', models TEXT NOT NULL DEFAULT '', - downshift TEXT, + prewalk TEXT, role TEXT NOT NULL DEFAULT '', note TEXT NOT NULL DEFAULT '', label TEXT NOT NULL DEFAULT '', @@ -139,6 +146,40 @@ CREATE TABLE IF NOT EXISTS experiments ( /** Directory names inside the jobs root that are not Harbor job dirs. */ const NON_JOB_DIRS = new Set(["_bench", "_manager"]); +/** True when a bun:sqlite error is a transient busy/recovery lock. */ +function isBusyLock(err: unknown): boolean { + if (err && typeof err === "object" && "code" in err) { + const code = err.code; + return typeof code === "string" && code.startsWith("SQLITE_BUSY"); + } + return false; +} + +/** + * Enable WAL journaling, tolerating a briefly locked database. + * + * `PRAGMA journal_mode = WAL` needs a momentary exclusive lock. When another + * connection holds the DB — a restarting manager, or a WAL mid-recovery — + * SQLite returns `SQLITE_BUSY`/`SQLITE_BUSY_RECOVERY`. The busy handler that + * `busy_timeout` installs is not invoked for recovery locks, so retry the + * pragma explicitly before surfacing the failure. + */ +function enableWal(db: Database): void { + const attempts = 10; + for (let attempt = 1; attempt <= attempts; attempt++) { + try { + db.run("PRAGMA journal_mode = WAL"); + return; + } catch (err) { + if (attempt < attempts && isBusyLock(err)) { + Bun.sleepSync(100); + continue; + } + throw err; + } + } +} + export class RunStore { #db: Database; readonly jobsDir: string; @@ -146,8 +187,9 @@ export class RunStore { constructor(jobsDir: string, dbPath?: string) { this.jobsDir = jobsDir; fs.mkdirSync(path.join(jobsDir, "_manager"), { recursive: true }); - this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite")); - this.#db.run("PRAGMA journal_mode = WAL"); + this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "metaharness.sqlite")); + this.#db.run("PRAGMA busy_timeout = 5000"); + enableWal(this.#db); this.#db.run(SCHEMA); const runColumns = new Set( (this.#db.query("PRAGMA table_info(runs)").all() as Array<{ name: string }>).map(c => c.name), @@ -165,11 +207,11 @@ export class RunStore { if (!runColumns.has("metrics_json")) { this.#db.run("ALTER TABLE runs ADD COLUMN metrics_json TEXT NOT NULL DEFAULT '{}'"); } - if (runColumns.has("slide") && !runColumns.has("downshift")) { - this.#db.run("ALTER TABLE runs RENAME COLUMN slide TO downshift"); + if (runColumns.has("slide") && !runColumns.has("prewalk")) { + this.#db.run("ALTER TABLE runs RENAME COLUMN slide TO prewalk"); } - if (!runColumns.has("slide") && !runColumns.has("downshift")) { - this.#db.run("ALTER TABLE runs ADD COLUMN downshift TEXT"); + if (!runColumns.has("slide") && !runColumns.has("prewalk")) { + this.#db.run("ALTER TABLE runs ADD COLUMN prewalk TEXT"); } const traceColumns = new Set( (this.#db.query("PRAGMA table_info(trials)").all() as Array<{ name: string }>).map(c => c.name), @@ -187,7 +229,7 @@ export class RunStore { this.#db .query( `INSERT INTO runs - (job_name, benchmark, dataset, agent, models, downshift, role, note, config_json, status, pid, created_at) + (job_name, benchmark, dataset, agent, models, prewalk, role, note, config_json, status, pid, created_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'running', ?, ?) ON CONFLICT(job_name) DO UPDATE SET benchmark = excluded.benchmark, pid = excluded.pid, status = 'running', @@ -201,7 +243,7 @@ export class RunStore { launch.dataset, launch.agent, launch.models.join(","), - launch.downshift ? JSON.stringify(launch.downshift) : null, + launch.prewalk ? JSON.stringify(launch.prewalk) : null, launch.role ?? "", launch.note ?? "", JSON.stringify(launch.config ?? {}), @@ -223,9 +265,39 @@ export class RunStore { .run(id, goal, Date.now()); } - getExperimentGoal(id: string): string { - const row = this.#db.query("SELECT goal FROM experiments WHERE id = ?").get(id) as { goal: string } | null; - return row?.goal ?? ""; + /** Stored experiment metadata, or null when the id was never registered. */ + getExperimentMeta(id: string): ExperimentMeta | null { + const row = this.#db.query("SELECT id, goal, updated_at FROM experiments WHERE id = ?").get(id) as { + id: string; + goal: string; + updated_at: number; + } | null; + return row ? { id: row.id, goal: row.goal, updatedAt: row.updated_at } : null; + } + + /** Every registered experiment row, newest first. */ + listExperimentMeta(): ExperimentMeta[] { + const rows = this.#db + .query("SELECT id, goal, updated_at FROM experiments ORDER BY updated_at DESC") + .all() as Array<{ + id: string; + goal: string; + updated_at: number; + }>; + return rows.map(r => ({ id: r.id, goal: r.goal, updatedAt: r.updated_at })); + } + + /** Drop the experiment metadata row (run rows are deleted separately via deleteRun). */ + deleteExperimentMeta(id: string): void { + this.#db.query("DELETE FROM experiments WHERE id = ?").run(id); + } + + /** Delete a run row and its trials; returns false when the run is unknown. */ + deleteRun(jobName: string): boolean { + if (!this.getRun(jobName)) return false; + this.#db.query("DELETE FROM trials WHERE job_name = ?").run(jobName); + this.#db.query("DELETE FROM runs WHERE job_name = ?").run(jobName); + return true; } /** Set role/note/label metadata on an existing run row. */ @@ -296,6 +368,15 @@ export class RunStore { trace_path = excluded.trace_path, updated_at = excluded.updated_at`, ); const tx = this.#db.transaction(() => { + // Prune rows whose trial dirs vanished from disk (a resume deletes + // interrupted trial dirs and re-runs the task under a fresh suffix) — + // otherwise phantom `running` rows haunt the dashboard forever. + if (snapshot.traces.length > 0) { + const names = snapshot.traces.map(t => t.name); + this.#db + .query(`DELETE FROM trials WHERE job_name = ? AND name NOT IN (${names.map(() => "?").join(",")})`) + .run(jobName, ...names); + } for (const trace of snapshot.traces) { upsert.run( jobName, @@ -331,10 +412,12 @@ export class RunStore { JSON.stringify(snapshot.metrics), jobName, ); - // Historical Harbor runs have no owning process. Infer their terminal - // state from result metadata or directory freshness. - if (row.benchmark === "harbor" && row.pid === null && row.finishedAt === null && row.status !== "cancelled") { - const result = readJobResult(jobDir); + // Runs with no owning process (historical dirs, or a runner that died + // with a previous manager). Infer terminal state from result metadata + // or directory freshness — an orphaned harbor child may still be + // running and writing trials, so a fresh dir stays "running". + if (row.pid === null && row.finishedAt === null && row.status !== "cancelled") { + const result = row.benchmark === "harbor" ? readJobResult(jobDir) : null; let status: RunStatus; let finishedAt: number | null = null; if (result?.finishedAt != null) { @@ -364,11 +447,13 @@ export class RunStore { }>; const out: RunRow[] = []; for (const { job_name } of active) { - // A pid-owning run whose process died without markExit (manager restart) - // is finalized here so it doesn't stay "running" forever. + // A pid-owning run whose runner died without markExit (manager + // restart) loses its pid here; syncRun's disk inference then decides + // the real status — the workload may have completed, or may still be + // running as an orphan. const row = this.getRun(job_name); if (row?.pid != null && !processAlive(row.pid)) { - this.markExit(job_name, null); + this.#db.query("UPDATE runs SET pid = NULL WHERE job_name = ?").run(job_name); } const synced = this.syncRun(job_name); if (synced) out.push(synced); @@ -424,7 +509,7 @@ function rowToRun(r: Record): RunRow { dataset: String(r.dataset), agent: String(r.agent), models: String(r.models), - downshift: r.downshift === null ? null : String(r.downshift), + prewalk: r.prewalk === null ? null : String(r.prewalk), config: JSON.parse(String(r.config_json ?? "{}")), role: String(r.role ?? "") as RunRole, note: String(r.note ?? ""), diff --git a/packages/harbor-manager/src/web/app.tsx b/packages/metaharness/src/web/app.tsx similarity index 97% rename from packages/harbor-manager/src/web/app.tsx rename to packages/metaharness/src/web/app.tsx index d951f4719..3afe3a0be 100644 --- a/packages/harbor-manager/src/web/app.tsx +++ b/packages/metaharness/src/web/app.tsx @@ -1,5 +1,5 @@ /** - * harbor-manager dashboard. + * metaharness dashboard. * * Views (hash-routed): * #/ experiments index — runs grouped by job-name prefix @@ -22,7 +22,7 @@ interface RunRow { dataset: string; agent: string; models: string; - downshift: string | null; + prewalk: string | null; config: Record; role: RunRole; note: string; @@ -747,7 +747,7 @@ const CELL_CLASS: Record = { /** * The comparison anchor for an experiment: the completed baseline arm with the - * highest pass rate (the "ceiling" a downshift arm tries to preserve). Ties + * highest pass rate (the "ceiling" a prewalk arm tries to preserve). Ties * break toward the cheaper arm. Returns null when no baseline has finished data. */ function pickReferenceArm(arms: ArmSummary[]): ArmSummary | null { @@ -799,7 +799,7 @@ function Delta({ /** * Launch a new arm into an existing experiment. The server inherits the * experiment's dataset and exact task sample from a sibling arm, so only the - * arm-specific knobs (name, model, role, note, optional downshift) are collected here. + * arm-specific knobs (name, model, role, note, optional prewalk) are collected here. */ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () => void }) { const [msg, setMsg] = useState(""); @@ -810,8 +810,8 @@ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () const body: Record = { arm: f.get("arm"), model: f.get("model") }; if (f.get("role")) body.role = f.get("role"); if (f.get("note")) body.note = f.get("note"); - if (f.get("downshiftInto") || f.get("downshift")) { - body.downshift = f.get("downshiftInto") ? { into: f.get("downshiftInto") } : {}; + if (f.get("prewalkInto") || f.get("prewalk")) { + body.prewalk = f.get("prewalkInto") ? { into: f.get("prewalkInto") } : {}; } setMsg("launching…"); const res = await fetch(`/api/experiments/${encodeURIComponent(experimentId)}/arms`, { @@ -838,9 +838,9 @@ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () - +
+ ) : ( + r.benchmark === "harbor" && + (r.done < r.nTotal || r.error > 0) && ( + + ) )} @@ -1769,7 +1788,7 @@ function RunsPage({ selected }: { selected: string | null }) { {detail.run.benchmark} · {detail.run.dataset} · {detail.run.models} {detail.run.score !== null ? ` · score ${(100 * detail.run.score).toFixed(1)}%` : ""} - {detail.run.downshift ? ` → ${detail.run.downshift}` : ""} + {detail.run.prewalk ? ` → ${detail.run.prewalk}` : ""}
{Object.entries(detail.run.metrics).map(([key, value]) => ( @@ -1861,8 +1880,8 @@ function LaunchForm({ onDone }: { onDone: () => void }) { if (f.get("goal")) body.goal = f.get("goal"); if (f.get("role")) body.role = f.get("role"); if (f.get("note")) body.note = f.get("note"); - if (f.get("downshiftInto") || f.get("downshift")) { - body.downshift = f.get("downshiftInto") ? { into: f.get("downshiftInto") } : {}; + if (f.get("prewalkInto") || f.get("prewalk")) { + body.prewalk = f.get("prewalkInto") ? { into: f.get("prewalkInto") } : {}; } setMsg("launching…"); const res = await fetch("/api/runs", { @@ -1890,9 +1909,9 @@ function LaunchForm({ onDone }: { onDone: () => void }) { - + @@ -1906,7 +1925,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) { - +