From e90bbf1218720e34af9c96921f0eb382811abb3a Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 14:49:44 +0200 Subject: [PATCH 01/28] feat(harbor-manager): replaced vite with native bun server bundling - Removed Vite and the dedicated dev script in favor of native Bun HTML bundling and HMR. - Updated database initialization to support robust WAL mode activation with retry logic for busy locks. - Improved task scheduling logic to ensure task count correctly aligns with explicit inclusion lists. - Adjusted provider derivation to treat explicit provider flags as authoritative overrides. --- packages/harbor-manager/package.json | 6 ++-- packages/harbor-manager/scripts/dev.ts | 39 ---------------------- packages/harbor-manager/src/runner.test.ts | 18 ++++++---- packages/harbor-manager/src/runner.ts | 8 ++++- packages/harbor-manager/src/server.ts | 37 ++++++-------------- packages/harbor-manager/src/store.ts | 37 +++++++++++++++++++- packages/harbor-manager/src/web/index.html | 2 +- packages/harbor-manager/vite.config.ts | 24 ------------- 8 files changed, 67 insertions(+), 104 deletions(-) delete mode 100755 packages/harbor-manager/scripts/dev.ts delete mode 100644 packages/harbor-manager/vite.config.ts diff --git a/packages/harbor-manager/package.json b/packages/harbor-manager/package.json index d6e7af6e7..1220e4750 100644 --- a/packages/harbor-manager/package.json +++ b/packages/harbor-manager/package.json @@ -21,7 +21,7 @@ "lint": "biome lint .", "start": "bun run src/server.ts", "serve": "bun run src/server.ts", - "dev": "bun scripts/dev.ts", + "dev": "bun --hot src/server.ts", "test": "bun test" }, "dependencies": { @@ -45,9 +45,7 @@ "@types/d3-scale": "^4.0.9", "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", - "@types/react-dom": "^19.1.0", - "@vitejs/plugin-react": "^5.0.4", - "vite": "catalog:" + "@types/react-dom": "^19.1.0" }, "engines": { "bun": ">=1.3.14" diff --git a/packages/harbor-manager/scripts/dev.ts b/packages/harbor-manager/scripts/dev.ts deleted file mode 100755 index 5cd7c102b..000000000 --- a/packages/harbor-manager/scripts/dev.ts +++ /dev/null @@ -1,39 +0,0 @@ -#!/usr/bin/env bun -/** - * Dev harness: runs the Bun API server (auto-restarting on server edits via - * `--watch`) and a Vite dev server (React Fast Refresh for the dashboard) - * together, tearing both down on one Ctrl-C. Vite proxies `/api` to the API - * server; the shared port travels through `HARBOR_API_PORT`. - * - * Extra args pass through to the API server: - * bun run dev -- --port 4700 --jobs-dir ../../runs/harbor - * - * Vite runs under Node (its bin shebang), the API under Bun; only the frontend - * hot-reloads in place, while server-side changes trigger a fast `--watch` restart. - */ -const args = Bun.argv.slice(2); -const portIndex = args.indexOf("--port"); -const apiPort = portIndex >= 0 ? (args[portIndex + 1] ?? "4700") : "4700"; -process.env.HARBOR_API_PORT = apiPort; - -const io = { stdout: "inherit", stderr: "inherit", stdin: "inherit", env: { ...process.env } } as const; -const api = Bun.spawn(["bun", "--watch", "src/server.ts", ...args], io); -const web = Bun.spawn(["vite"], io); - -let stopping = false; -const stop = (): void => { - if (stopping) return; - stopping = true; - try { - api.kill(); - } catch {} - try { - web.kill(); - } catch {} -}; -process.on("SIGINT", stop); -process.on("SIGTERM", stop); - -await Promise.race([api.exited, web.exited]); -stop(); -process.exit(0); diff --git a/packages/harbor-manager/src/runner.test.ts b/packages/harbor-manager/src/runner.test.ts index 1b5e53431..922e74de0 100644 --- a/packages/harbor-manager/src/runner.test.ts +++ b/packages/harbor-manager/src/runner.test.ts @@ -25,13 +25,17 @@ describe("generic agent-arg / env passthrough", () => { expect(env.OMP_BENCH_AGENT_ARGS).toBeUndefined(); }); - it("routes an explicit --providers entry alongside the model's own provider", () => { - // The runner has no built-in concept of a "second model"; gateway routing - // for any extra model introduced via --agent-arg must be declared - // explicitly via --providers. - const cfg = parseArgs(["--model", "anthropic/claude-opus-4-8", "--providers", "google"]); - const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); - expect(new Set(env.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["anthropic", "google"])); + it("explicit --providers is authoritative; the default derives from the model", () => { + // Explicit list: exactly what was asked for — the escape hatch that lets + // the model's own provider authenticate directly (forwarded env key) + // while only e.g. oauth-only providers route through the gateway. + const explicit = parseArgs(["--model", "anthropic/claude-opus-4-8", "--providers", "google"]); + const envExplicit = buildHarborEnv(explicit, "/tmp/models.yml", null, "test"); + expect(new Set(envExplicit.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["google"])); + // No flag: the model's provider is gateway-routed by default. + const derived = parseArgs(["--model", "anthropic/claude-opus-4-8"]); + const envDerived = buildHarborEnv(derived, "/tmp/models.yml", null, "test"); + expect(new Set(envDerived.OMP_BENCH_GATEWAY_PROVIDERS?.split(","))).toEqual(new Set(["anthropic"])); }); it("collects explicit --env pairs, with an explicit value winning over a bare host-forwarded key", () => { diff --git a/packages/harbor-manager/src/runner.ts b/packages/harbor-manager/src/runner.ts index 8b7d2b362..89e95e215 100755 --- a/packages/harbor-manager/src/runner.ts +++ b/packages/harbor-manager/src/runner.ts @@ -1059,7 +1059,13 @@ function buildMountsJson(source: SourceMount | null): string | null { } function deriveProviders(cfg: Config): string[] { - const set = new Set(cfg.providers); + // Explicit --providers is authoritative: it's the escape hatch for routing + // only SOME providers through the gateway (e.g. oauth-only openai-codex) + // while the model's own provider authenticates directly via a forwarded + // env key. The model-provider + anthropic/openai-codex additions are the + // DEFAULT for when the flag is absent. + if (cfg.providers.length > 0) return [...new Set(cfg.providers)]; + const set = new Set(); for (const m of cfg.models) { const slash = m.indexOf("/"); if (slash > 0) set.add(m.slice(0, slash)); diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index 7ccca4368..edf24119f 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -27,7 +27,7 @@ export interface ExperimentMetaUpdate { runs?: Record; } -const INDEX_HTML_PATH = new URL("./web/index.html", import.meta.url).pathname; +import indexHtml from "./web/index.html"; const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); const PKG_DIR = path.resolve(import.meta.dir, ".."); @@ -188,7 +188,6 @@ export class ManagerServer { #lastSnapshot = ""; #syncTimer: Timer | undefined; #server: Server | null = null; - #appBundleCode: string | null = null; readonly jobsDir: string; constructor(jobsDir: string, dbPath?: string) { @@ -207,6 +206,10 @@ export class ManagerServer { this.#server = Bun.serve({ port, idleTimeout: 0, + // Bun bundles the dashboard (React + TSX) from the HTML import and + // serves it on the same port as the API — one process, no Vite. + routes: { "/": indexHtml }, + development: process.env.NODE_ENV !== "production" && { hmr: true, console: true }, fetch: request => this.#route(request), }); return this.#server; @@ -234,22 +237,6 @@ export class ManagerServer { } } - /** Bundle the React dashboard once per process; served at /app.tsx (matches the Vite dev entry). */ - async #appBundle(): Promise { - if (this.#appBundleCode !== null) return this.#appBundleCode; - const result = await Bun.build({ - entrypoints: [path.join(import.meta.dir, "web", "app.tsx")], - target: "browser", - minify: true, - define: { "process.env.NODE_ENV": '"production"' }, - }); - if (!result.success) { - throw new Error(`dashboard bundle failed:\n${result.logs.map(l => l.message).join("\n")}`); - } - this.#appBundleCode = await result.outputs[0].text(); - return this.#appBundleCode; - } - #broadcast(frame: string): void { const bytes = new TextEncoder().encode(frame); for (const client of this.#sse) { @@ -267,14 +254,6 @@ export class ManagerServer { const url = new URL(request.url); const p = url.pathname; try { - if (p === "/" || p === "/index.html") { - return new Response(Bun.file(INDEX_HTML_PATH)); - } - if (p === "/app.tsx") { - return new Response(await this.#appBundle(), { - headers: { "content-type": "text/javascript; charset=utf-8" }, - }); - } if (p === "/api/events") return this.#sseResponse(); if (p === "/api/benchmarks" && request.method === "GET") { return Response.json(BENCHMARK_DEFINITIONS); @@ -402,7 +381,11 @@ export class ManagerServer { this.jobsDir, ]; if (request.agent) argv.push("--agent", request.agent); - if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks)); + // An explicit include list IS the sample — never let the runner's + // default task cap truncate it. + const tasks = + request.tasks ?? (request.include && request.include.length > 0 ? request.include.length : undefined); + if (tasks !== undefined) argv.push("--tasks", String(tasks)); if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency)); if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts)); if (request.timeoutMultiplier !== undefined) diff --git a/packages/harbor-manager/src/store.ts b/packages/harbor-manager/src/store.ts index cf3653375..960338357 100644 --- a/packages/harbor-manager/src/store.ts +++ b/packages/harbor-manager/src/store.ts @@ -139,6 +139,40 @@ CREATE TABLE IF NOT EXISTS experiments ( /** Directory names inside the jobs root that are not Harbor job dirs. */ const NON_JOB_DIRS = new Set(["_bench", "_manager"]); +/** True when a bun:sqlite error is a transient busy/recovery lock. */ +function isBusyLock(err: unknown): boolean { + if (err && typeof err === "object" && "code" in err) { + const code = err.code; + return typeof code === "string" && code.startsWith("SQLITE_BUSY"); + } + return false; +} + +/** + * Enable WAL journaling, tolerating a briefly locked database. + * + * `PRAGMA journal_mode = WAL` needs a momentary exclusive lock. When another + * connection holds the DB — a restarting manager, or a WAL mid-recovery — + * SQLite returns `SQLITE_BUSY`/`SQLITE_BUSY_RECOVERY`. The busy handler that + * `busy_timeout` installs is not invoked for recovery locks, so retry the + * pragma explicitly before surfacing the failure. + */ +function enableWal(db: Database): void { + const attempts = 10; + for (let attempt = 1; attempt <= attempts; attempt++) { + try { + db.run("PRAGMA journal_mode = WAL"); + return; + } catch (err) { + if (attempt < attempts && isBusyLock(err)) { + Bun.sleepSync(100); + continue; + } + throw err; + } + } +} + export class RunStore { #db: Database; readonly jobsDir: string; @@ -147,7 +181,8 @@ export class RunStore { this.jobsDir = jobsDir; fs.mkdirSync(path.join(jobsDir, "_manager"), { recursive: true }); this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite")); - this.#db.run("PRAGMA journal_mode = WAL"); + this.#db.run("PRAGMA busy_timeout = 5000"); + enableWal(this.#db); this.#db.run(SCHEMA); const runColumns = new Set( (this.#db.query("PRAGMA table_info(runs)").all() as Array<{ name: string }>).map(c => c.name), diff --git a/packages/harbor-manager/src/web/index.html b/packages/harbor-manager/src/web/index.html index 857a76165..70e56e597 100644 --- a/packages/harbor-manager/src/web/index.html +++ b/packages/harbor-manager/src/web/index.html @@ -11,6 +11,6 @@
- + diff --git a/packages/harbor-manager/vite.config.ts b/packages/harbor-manager/vite.config.ts deleted file mode 100644 index 063604c6e..000000000 --- a/packages/harbor-manager/vite.config.ts +++ /dev/null @@ -1,24 +0,0 @@ -import react from "@vitejs/plugin-react"; -import { defineConfig } from "vite"; - -/** - * Dev-only Vite config for the harbor-manager dashboard. - * - * Serves `src/web` with React Fast Refresh and proxies the API — including the - * SSE run stream at `/api/events` — to the Bun server that `scripts/dev.ts` - * starts alongside it (`HARBOR_API_PORT`, default 4700). Production is served by - * the Bun server itself (`src/server.ts` bundles `app.tsx` on demand); Vite is - * not part of the production path. - */ -const apiTarget = `http://localhost:${process.env.HARBOR_API_PORT ?? "4700"}`; - -export default defineConfig({ - root: "src/web", - server: { - port: Number(process.env.HARBOR_WEB_PORT ?? "5173"), - proxy: { - "/api": { target: apiTarget, changeOrigin: true }, - }, - }, - plugins: [react()], -}); From 4df6f6683d1ab50d6e8b0288b2533f5fb1272814 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 15:18:31 +0200 Subject: [PATCH 02/28] chore: finalizing the new /prewalk --- packages/coding-agent/CHANGELOG.md | 10 +- packages/coding-agent/src/cli/args.ts | 14 +- packages/coding-agent/src/cli/flag-tables.ts | 8 +- packages/coding-agent/src/commands/launch.ts | 12 +- .../src/config/settings-schema.ts | 10 +- packages/coding-agent/src/main.ts | 18 +-- ...hift-checklist.md => prewalk-checklist.md} | 0 ...nshift-continue.md => prewalk-continue.md} | 0 .../{downshift-plan.md => prewalk-plan.md} | 0 packages/coding-agent/src/sdk.ts | 8 +- .../coding-agent/src/session/agent-session.ts | 122 +++++++++--------- .../src/slash-commands/builtin-registry.ts | 10 +- ....test.ts => agent-session-prewalk.test.ts} | 26 ++-- packages/harbor-manager/README.md | 2 +- .../harbor-manager/src/experiments.test.ts | 8 +- packages/harbor-manager/src/experiments.ts | 10 +- packages/harbor-manager/src/manager.test.ts | 14 +- packages/harbor-manager/src/runner.test.ts | 6 +- packages/harbor-manager/src/server.ts | 24 ++-- packages/harbor-manager/src/store.ts | 24 ++-- packages/harbor-manager/src/web/app.tsx | 26 ++-- 21 files changed, 176 insertions(+), 176 deletions(-) rename packages/coding-agent/src/prompts/system/{downshift-checklist.md => prewalk-checklist.md} (100%) rename packages/coding-agent/src/prompts/system/{downshift-continue.md => prewalk-continue.md} (100%) rename packages/coding-agent/src/prompts/system/{downshift-plan.md => prewalk-plan.md} (100%) rename packages/coding-agent/test/{agent-session-downshift.test.ts => agent-session-prewalk.test.ts} (94%) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9fe2494f3..f93c9c872 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,11 +4,11 @@ ### Breaking Changes -- Replaced the `--reasoning-slide-*` flag family (`--reasoning-slide-model`, `--reasoning-slide-turns`, `--reasoning-slide-on-action`, `--reasoning-slide-plan`, `--reasoning-slide-plan-at`, `--reasoning-slide-checklist`) with a single downshift mechanism: `--downshift` switches from the starting model to a fast/cheap target at the first completed turn that starts execution — the todo-list init the plan nudge asks for, or any edit/write tool — always with the hidden plan nudge before the switch and the verify-before-finishing checklist after it (the configuration that won benchmark testing). `--downshift-into ` overrides the default "smol"-role target and implies `--downshift`; `--no-downshift` force-disables. The fixed-turn trigger and per-piece plan/checklist toggles are gone. +- Replaced the `--reasoning-slide-*` flag family (`--reasoning-slide-model`, `--reasoning-slide-turns`, `--reasoning-slide-on-action`, `--reasoning-slide-plan`, `--reasoning-slide-plan-at`, `--reasoning-slide-checklist`) with a single prewalk mechanism: `--prewalk` switches from the starting model to a fast/cheap target at the first completed turn that starts execution — the todo-list init the plan nudge asks for, or any edit/write tool — always with the hidden plan nudge before the switch and the verify-before-finishing checklist after it (the configuration that won benchmark testing). `--prewalk-into ` overrides the default "smol"-role target and implies `--prewalk`; `--no-prewalk` force-disables. The fixed-turn trigger and per-piece plan/checklist toggles are gone. ### Added -- Added `--downshift` / `--downshift-into ` / `--no-downshift`: start on a strong model, then hand off to a fast/cheap one (default the `smol` role) at the first edit/write tool call *after* the todo list has been initialized. The starting model handles all planning and todo initialization, and begins implementation, before handing off; the fast model includes a verify checklist before finishing. Enable per-user with the `downshift.enabled` setting; force it mid-session with the new `/downshift` slash command, which arms the switch. +- Added `--prewalk` / `--prewalk-into ` / `--no-prewalk`: start on a strong model, then hand off to a fast/cheap one (default the `smol` role) at the first edit/write tool call *after* the todo list has been initialized. The starting model handles all planning and todo initialization, and begins implementation, before handing off; the fast model includes a verify checklist before finishing. Enable per-user with the `prewalk.enabled` setting; force it mid-session with the new `/prewalk` slash command, which arms the switch. - Added display setting to toggle between collapsing or keeping compacted history inline, now applied to live session displays - Added a compact session-only model picker (Alt+P) for quick model switching without changing roles - Added `@` search to the Alt+P / `/switch` picker: it lists configured Ctrl+P quick roles in matching segment colors and applies the selected role's model and thinking for the current session. @@ -18,7 +18,7 @@ ### Changed -- Refined the downshift planning instructions to require a super-detailed todo list with one item per concrete step, each specifying its target and verification method +- Refined the prewalk planning instructions to require a super-detailed todo list with one item per concrete step, each specifying its target and verification method - Updated tangential agent forks to ignore parent session history and focus exclusively on the new request - Hardened `/tan` fork isolation: the clone's inherited todo list is cleared at fork (parent todo reminders no longer drag the tan back onto the parent's task), the fork notice warns that the parent is concurrently editing the same working directory, and the notice is re-injected after each compaction so the fork boundary survives summarization - Added visual markers in the transcript for elided tool calls that have no corresponding result @@ -27,7 +27,7 @@ ### Removed -- Removed the `--downshift-boomerang` feature and its associated configuration setting +- Removed the `--prewalk-boomerang` feature and its associated configuration setting - Removed the unreliable Bing and Yahoo HTML-scraping web search providers ### Fixed @@ -39,7 +39,7 @@ - Fixed transcript rebuilds (compaction, `/compact`, and toggling history display) repainting content below stale scrollback when collapsing history; rebuilds now correctly clear the scrollback buffer when history is collapsed - Improved auto-compaction to automatically drop images and elide content when context is tight, and added persistent warning badges to the compaction divider when manual intervention is required - Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately -- Fixed the downshift plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls +- Fixed the prewalk plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls - Fixed launch tool rendering stacking a stale pending header over a bare `✓ Launch` line and raw text: the tool now uses a merged registry renderer with one per-op status header (op, target, `state · pid · uptime` meta), stripped log cursor suffixes, capped collapsed log/list previews, and a launch tool glyph - Fixed confusing launch start/wait results when readiness timed out with the log pattern already matched (readiness needs log AND port): the result printed a contradictory `Ready: ` next to `Readiness timed out` without naming the failing condition. Daemon snapshots now carry the unmet conditions (`readyPending`), and start/wait results state exactly what never happened (e.g. `port 3100 on 127.0.0.1 never accepted connections`); the TUI shows a `waiting on port` badge on starting daemons - Fixed the in-process `stat` builtin mangling BSD-style invocations like `stat -f "%Sm %N" file` (macOS muscle memory): GNU `-f` means `--file-system`, so the format string was treated as a file operand — printing filesystem info for the real operands and erroring with `cannot read file system information for '%Sm %N'`. A `-f` whose format value contains `%` is now detected as BSD syntax and translated to the GNU equivalent (`%Sm`→`%y`, `%N`→`%n`, `%z`→`%s`, epoch/`S`-form times, owner/group/permission and `H`/`L` sub-field directives, `-L`/`-n`/`-q`/`-F` flag clusters, with `%n`/`%t` as literal newline/tab); directives with no GNU counterpart fail with a clear `unsupported BSD format directive` error diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index d60d3a9d3..f0a1c0195 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -27,9 +27,9 @@ export interface Args { smol?: string; slow?: string; plan?: string; - downshift?: boolean; - noDownshift?: boolean; - downshiftInto?: string; + prewalk?: boolean; + noPrewalk?: boolean; + prewalkInto?: string; planYolo?: boolean; planYoloInto?: string; maxTime?: number; @@ -236,10 +236,10 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map = { "--plan": (result, value) => { result.plan = value; }, - "--downshift-into": (result, value) => { - result.downshiftInto = value; + "--prewalk-into": (result, value) => { + result.prewalkInto = value; }, "--plan-yolo-into": (result, value) => { result.planYoloInto = value; @@ -281,8 +281,8 @@ export const VALUELESS_FLAGS: ReadonlySet = new Set([ "--no-pty", "--hide-thinking", "--advisor", - "--downshift", - "--no-downshift", + "--prewalk", + "--no-prewalk", "--plan-yolo", "--print", "--print-thoughts", diff --git a/packages/coding-agent/src/commands/launch.ts b/packages/coding-agent/src/commands/launch.ts index 915d56971..49f4df8ee 100644 --- a/packages/coding-agent/src/commands/launch.ts +++ b/packages/coding-agent/src/commands/launch.ts @@ -34,15 +34,15 @@ export default class Index extends Command { plan: Flags.string({ description: "Plan model for architectural planning (or PI_PLAN_MODEL env)", }), - downshift: Flags.boolean({ + prewalk: Flags.boolean({ description: - "Switch from the active model to a fast/cheap model at the first edit/write after the plan's todo list exists (default off; see downshift.enabled)", + "Switch from the active model to a fast/cheap model at the first edit/write after the plan's todo list exists (default off; see prewalk.enabled)", }), - "no-downshift": Flags.boolean({ - description: "Disable downshift even if downshift.enabled is set", + "no-prewalk": Flags.boolean({ + description: "Disable prewalk even if prewalk.enabled is set", }), - "downshift-into": Flags.string({ - description: 'Target model for downshift (default the "smol" role)', + "prewalk-into": Flags.string({ + description: 'Target model for prewalk (default the "smol" role)', }), "plan-yolo": Flags.boolean({ description: diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 5888b7ce0..468780b61 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -117,7 +117,7 @@ export const TAB_METADATA: Record = { appearance: ["Theme", "Status Line", "Display", "Images"], - model: ["Thinking", "Sampling", "Prompt", "Retry & Fallback", "Advisor", "Downshift", "Vision"], + model: ["Thinking", "Sampling", "Prompt", "Retry & Fallback", "Advisor", "Prewalk", "Vision"], interaction: [ "Input", "Approvals", @@ -422,15 +422,15 @@ export const SETTINGS_SCHEMA = { "Pair a second model (assigned to the 'advisor' role) that passively reviews each turn and injects notes.", }, }, - "downshift.enabled": { + "prewalk.enabled": { type: "boolean", default: false, ui: { tab: "model", - group: "Downshift", - label: "Enable Downshift", + group: "Prewalk", + label: "Enable Prewalk", description: - "Start on the active model, then switch to a fast/cheap model (default the 'smol' role) at the first edit/write after the plan nudge's todo list exists — the strong model plans, commits the todos, and starts the implementation before handing off. Overridable per session with --downshift / --no-downshift.", + "Start on the active model, then switch to a fast/cheap model (default the 'smol' role) at the first edit/write after the plan nudge's todo list exists — the strong model plans, commits the todos, and starts the implementation before handing off. Overridable per session with --prewalk / --no-prewalk.", }, }, "advisor.subagents": { diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index f91f1e31c..41943669f 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -905,27 +905,27 @@ export async function buildSessionOptions( if (!options.model) options.model = scopedModels[0].model; } - if (parsed.noDownshift && (parsed.downshift || parsed.downshiftInto !== undefined)) { - throw new Error("--no-downshift cannot be combined with --downshift or --downshift-into"); + if (parsed.noPrewalk && (parsed.prewalk || parsed.prewalkInto !== undefined)) { + throw new Error("--no-prewalk cannot be combined with --prewalk or --prewalk-into"); } - const downshiftEnabled = parsed.noDownshift + const prewalkEnabled = parsed.noPrewalk ? false - : parsed.downshift === true || parsed.downshiftInto !== undefined + : parsed.prewalk === true || parsed.prewalkInto !== undefined ? true - : activeSettings.get("downshift.enabled"); - if (downshiftEnabled) { - const rolePattern = expandRoleAlias(parsed.downshiftInto ?? "pi/smol", activeSettings); + : activeSettings.get("prewalk.enabled"); + if (prewalkEnabled) { + const rolePattern = expandRoleAlias(parsed.prewalkInto ?? "pi/smol", activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.downshiftInto ?? "pi/smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? "pi/smol"}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); } - options.downshift = { target: resolved.model, thinkingLevel: resolved.thinkingLevel }; + options.prewalk = { target: resolved.model, thinkingLevel: resolved.thinkingLevel }; } if (parsed.planYoloInto !== undefined && !parsed.planYolo) { diff --git a/packages/coding-agent/src/prompts/system/downshift-checklist.md b/packages/coding-agent/src/prompts/system/prewalk-checklist.md similarity index 100% rename from packages/coding-agent/src/prompts/system/downshift-checklist.md rename to packages/coding-agent/src/prompts/system/prewalk-checklist.md diff --git a/packages/coding-agent/src/prompts/system/downshift-continue.md b/packages/coding-agent/src/prompts/system/prewalk-continue.md similarity index 100% rename from packages/coding-agent/src/prompts/system/downshift-continue.md rename to packages/coding-agent/src/prompts/system/prewalk-continue.md diff --git a/packages/coding-agent/src/prompts/system/downshift-plan.md b/packages/coding-agent/src/prompts/system/prewalk-plan.md similarity index 100% rename from packages/coding-agent/src/prompts/system/downshift-plan.md rename to packages/coding-agent/src/prompts/system/prewalk-plan.md diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index b6abff059..08d3eee01 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -109,7 +109,7 @@ import { obfuscateProviderContext, SecretObfuscator, } from "./secrets"; -import { AgentSession, type Downshift, type PlanYolo } from "./session/agent-session"; +import { AgentSession, type Prewalk, type PlanYolo } from "./session/agent-session"; import { discoverAuthStorage as discoverAuthStorageFromConfig } from "./session/auth-broker-config"; import type { AuthStorage } from "./session/auth-storage"; import { @@ -407,8 +407,8 @@ export interface CreateAgentSessionOptions { thinkingLevel?: ConfiguredThinkingLevel; /** Models available for cycling (Ctrl+P in interactive mode) */ scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; - /** Downshift from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ - downshift?: Downshift; + /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ + prewalk?: Prewalk; /** Force read-only plan mode at start, auto-approve on the model's first resolve call, then switch to execute. */ planYolo?: PlanYolo; @@ -2867,7 +2867,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} agent, pruneToolDescriptions: inlineToolDescriptors, thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel, - downshift: options.downshift, + prewalk: options.prewalk, planYolo: options.planYolo, serviceTierByFamily: initialServiceTierByFamily, sessionManager, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 6bf976fbc..68ac6c306 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -259,9 +259,9 @@ import goalModeContextPrompt from "../prompts/goals/goal-mode-context.md" with { import goalTodoContextPrompt from "../prompts/goals/goal-todo-context.md" with { type: "text" }; import parentIrcSteerTemplate from "../prompts/steering/parent-irc.md" with { type: "text" }; import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type: "text" }; -import downshiftChecklistPrompt from "../prompts/system/downshift-checklist.md" with { type: "text" }; -import downshiftContinuePrompt from "../prompts/system/downshift-continue.md" with { type: "text" }; -import downshiftPlanPrompt from "../prompts/system/downshift-plan.md" with { type: "text" }; +import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; +import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; +import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; @@ -419,29 +419,29 @@ const MID_RUN_TODO_NUDGE_MUTATING_TOOLS: Record = { /** `customType` for the hidden mid-run todo nudge; `display: false`, so it reaches * the model but never renders in the TUI or transcript. */ const MID_RUN_TODO_NUDGE_MESSAGE_TYPE = "mid-run-todo-nudge"; -/** Hidden plan nudge injected by downshift; scrubbed from the LLM context +/** Hidden plan nudge injected by prewalk; scrubbed from the LLM context * when the switch happens. */ -const DOWNSHIFT_PLAN_MESSAGE_TYPE = "downshift-plan"; +const PREWALK_PLAN_MESSAGE_TYPE = "prewalk-plan"; /** Hidden safety-net nudge forcing one more turn after a text-only reply to * the plan nudge, which would otherwise end the run with no code written. */ -const DOWNSHIFT_CONTINUE_MESSAGE_TYPE = "downshift-continue"; +const PREWALK_CONTINUE_MESSAGE_TYPE = "prewalk-continue"; /** Hidden "verify before finishing" checklist steered into the run at the * switch, aimed at the fast model's specific failure patterns: partial * multi-site fixes, unnecessarily broad rewrites, and reported-test-only * verification. */ -const DOWNSHIFT_CHECKLIST_MESSAGE_TYPE = "downshift-checklist"; +const PREWALK_CHECKLIST_MESSAGE_TYPE = "prewalk-checklist"; /** Tools whose first successful call triggers the switch — once the todo - * gate is open (see {@link AgentSession.#downshiftTodoSeen}). Bash is + * gate is open (see {@link AgentSession.#prewalkTodoSeen}). Bash is * deliberately excluded: it doubles as exploration (ls/cat) and fired * turn-1 switches in practice. `todo` is deliberately NOT a trigger: firing * at the todo init handed the fast model 100% of the implementation with * zero started work and measurably regressed pass rates. */ -const DOWNSHIFT_ACTION_TOOLS: Record = { +const PREWALK_ACTION_TOOLS: Record = { edit: true, write: true, }; /** `customType` for the hidden hand-off message steered to the target model - * once PlanYolo auto-approves the plan. Unlike downshift's plan nudge this + * once PlanYolo auto-approves the plan. Unlike prewalk's plan nudge this * is never scrubbed — it IS the instruction the target model acts on. */ const PLAN_YOLO_HANDOFF_MESSAGE_TYPE = "plan-yolo-handoff"; /** Abort reason for the Gemini reasoning-header runaway interrupt. Surfaced on the @@ -712,7 +712,7 @@ export interface AsyncJobSnapshot { export type { ShakeMode, ShakeResult }; /** - * Downshift: switches an active session one-way from its starting model to + * Prewalk: switches an active session one-way from its starting model to * a fast/cheap `target` at the first completed turn that runs an edit/write * tool once the todo list exists. A hidden plan nudge asks the starting * model to write a plan, initialize its todo list from it, and start; the @@ -722,7 +722,7 @@ export type { ShakeMode, ShakeResult }; * finishing. Both are always on — this is the one mechanism that won out * over turn-count and ungated variants in testing. */ -export interface Downshift { +export interface Prewalk { target: Model; thinkingLevel?: ConfiguredThinkingLevel; } @@ -754,8 +754,8 @@ export interface AgentSessionConfig { scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; /** Initial session thinking selector. */ thinkingLevel?: ConfiguredThinkingLevel; - /** Downshift from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ - downshift?: Downshift; + /** Prewalk from the starting model to a fast/cheap target at the first edit/write once the todo list exists. */ + prewalk?: Prewalk; /** Force read-only plan mode at start, auto-approve on the model's first * `resolve` call, then switch to the target to implement. */ planYolo?: PlanYolo; @@ -1672,13 +1672,13 @@ export class AgentSession { #autoThinking: boolean = false; /** The level `auto` last resolved to (for UI); undefined until a turn is classified. */ #autoResolvedLevel: Effort | undefined; - #downshift: Downshift | undefined; + #prewalk: Prewalk | undefined; /** True once the plan nudge has been queued; scrubbed from context at the switch. */ - #downshiftPlanInjected = false; - /** True once any successful `todo` call landed — opens the downshift + #prewalkPlanInjected = false; + /** True once any successful `todo` call landed — opens the prewalk * trigger gate: the switch fires at the first edit/write AFTER the todo * list exists (sessions without a todo tool skip the gate). */ - #downshiftTodoSeen = false; + #prewalkTodoSeen = false; #planYolo: PlanYolo | undefined; #planYoloPreviousTools: string[] | undefined; #planYoloArmed = false; @@ -2167,10 +2167,10 @@ export class AgentSession { this.#emit(pending); } - /** Advance the one-way downshift switch at a completed assistant-turn boundary. */ - async #advanceDownshift(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { - const downshift = this.#downshift; - if (!downshift || context?.message.role !== "assistant") return; + /** Advance the one-way prewalk switch at a completed assistant-turn boundary. */ + async #advancePrewalk(liveMessages: AgentMessage[], context: AgentTurnEndContext | undefined): Promise { + const prewalk = this.#prewalk; + if (!prewalk || context?.message.role !== "assistant") return; // Structural safety net: every branch below assumes the agent loop will // run another turn. It won't if THIS turn had no tool calls — the loop @@ -2180,11 +2180,11 @@ export class AgentSession { // silently killing production SWE-bench runs before any code was ever // written. Force one more turn only in that specific, self-created // hazard window. - if (this.#downshiftPlanInjected && context.toolResults.length === 0) { + if (this.#prewalkPlanInjected && context.toolResults.length === 0) { this.agent.steer({ role: "custom", - customType: DOWNSHIFT_CONTINUE_MESSAGE_TYPE, - content: downshiftContinuePrompt, + customType: PREWALK_CONTINUE_MESSAGE_TYPE, + content: prewalkContinuePrompt, attribution: "agent", display: false, timestamp: Date.now(), @@ -2198,24 +2198,24 @@ export class AgentSession { // the fast model the whole implementation cold. Sessions without a todo // tool skip the gate. if (context.toolResults.some(result => result.toolName === "todo")) { - this.#downshiftTodoSeen = true; + this.#prewalkTodoSeen = true; } - const todoGateOpen = this.#downshiftTodoSeen || !this.#toolRegistry.has("todo"); + const todoGateOpen = this.#prewalkTodoSeen || !this.#toolRegistry.has("todo"); const action = todoGateOpen - ? context.toolResults.find(result => DOWNSHIFT_ACTION_TOOLS[result.toolName]) + ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) : undefined; if (!action) { - if (!this.#downshiftPlanInjected) { - this.#downshiftPlanInjected = true; + if (!this.#prewalkPlanInjected) { + this.#prewalkPlanInjected = true; this.agent.steer({ role: "custom", - customType: DOWNSHIFT_PLAN_MESSAGE_TYPE, - content: downshiftPlanPrompt, + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, display: false, attribution: "agent", timestamp: Date.now(), }); - this.emitNotice("info", "Downshift: injected deep-plan nudge.", "downshift"); + this.emitNotice("info", "Prewalk: injected deep-plan nudge.", "prewalk"); } return; } @@ -2225,24 +2225,24 @@ export class AgentSession { await this.#waitForSessionMessagePersistence(toolResult); } - this.#scrubDownshiftPlanNudge(liveMessages); - const target = downshift.target; + this.#scrubPrewalkPlanNudge(liveMessages); + const target = prewalk.target; if (this.model && modelsAreEqual(this.model, target)) { - this.#downshift = undefined; + this.#prewalk = undefined; return; } - await this.setModelTemporary(target, downshift.thinkingLevel, { ephemeral: true }); - this.#downshift = undefined; + await this.setModelTemporary(target, prewalk.thinkingLevel, { ephemeral: true }); + this.#prewalk = undefined; this.emitNotice( "info", - `Downshift: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, - "downshift", + `Prewalk: switched to ${target.provider}/${target.id} after first ${action.toolName} call.`, + "prewalk", ); this.agent.steer({ role: "custom", - customType: DOWNSHIFT_CHECKLIST_MESSAGE_TYPE, - content: downshiftChecklistPrompt, + customType: PREWALK_CHECKLIST_MESSAGE_TYPE, + content: prewalkChecklistPrompt, attribution: "agent", display: false, timestamp: Date.now(), @@ -2250,35 +2250,35 @@ export class AgentSession { } /** - * Arm downshift outside the normal startup path (the `/downshift` slash + * Arm prewalk outside the normal startup path (the `/prewalk` slash * command): sets the target and immediately steers the plan nudge rather * than waiting for the next turn boundary, since an explicit manual - * invocation means "start this now." A no-op with a notice if a downshift + * invocation means "start this now." A no-op with a notice if a prewalk * is already armed and waiting. */ - armDownshift(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { - if (this.#downshift) { + armPrewalk(target: Model, thinkingLevel?: ConfiguredThinkingLevel): void { + if (this.#prewalk) { this.emitNotice( "info", - `Downshift: already armed for ${this.#downshift.target.provider}/${this.#downshift.target.id}, waiting for the first edit/write.`, - "downshift", + `Prewalk: already armed for ${this.#prewalk.target.provider}/${this.#prewalk.target.id}, waiting for the first edit/write.`, + "prewalk", ); return; } - this.#downshift = { target, thinkingLevel }; - this.#downshiftPlanInjected = true; + this.#prewalk = { target, thinkingLevel }; + this.#prewalkPlanInjected = true; this.agent.steer({ role: "custom", - customType: DOWNSHIFT_PLAN_MESSAGE_TYPE, - content: downshiftPlanPrompt, + customType: PREWALK_PLAN_MESSAGE_TYPE, + content: prewalkPlanPrompt, display: false, attribution: "agent", timestamp: Date.now(), }); this.emitNotice( "info", - `Downshift: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, - "downshift", + `Prewalk: armed for ${target.provider}/${target.id} — will switch at the first edit/write once the todo list exists.`, + "prewalk", ); } @@ -2288,12 +2288,12 @@ export class AgentSession { * Splices the loop's live context array in place (the run streams from * it) and mirrors the removal into agent state. The persisted transcript * keeps the message for audit; a session reload re-materializes it, - * which is acceptable for downshift's single-run lifecycle. + * which is acceptable for prewalk's single-run lifecycle. */ - #scrubDownshiftPlanNudge(liveMessages: AgentMessage[]): void { - if (!this.#downshiftPlanInjected) return; + #scrubPrewalkPlanNudge(liveMessages: AgentMessage[]): void { + if (!this.#prewalkPlanInjected) return; const isPlanNudge = (m: AgentMessage): boolean => - m.role === "custom" && m.customType === DOWNSHIFT_PLAN_MESSAGE_TYPE; + m.role === "custom" && m.customType === PREWALK_PLAN_MESSAGE_TYPE; for (let i = liveMessages.length - 1; i >= 0; i--) { if (isPlanNudge(liveMessages[i])) liveMessages.splice(i, 1); } @@ -2434,8 +2434,8 @@ export class AgentSession { } else { this.#thinkingLevel = config.thinkingLevel; } - if (config.downshift) { - this.#downshift = config.downshift; + if (config.prewalk) { + this.#prewalk = config.prewalk; } if (config.planYolo) { this.#planYolo = config.planYolo; @@ -2514,7 +2514,7 @@ export class AgentSession { }); if (detection) this.#maybeInjectToolCallLoopRedirect(messages, detection); } - await this.#advanceDownshift(messages, context); + await this.#advancePrewalk(messages, context); this.#advisorPrimaryTurnsCompleted++; if (this.#advisors.length > 0) { for (const a of this.#advisors) { diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 72ec7cfb9..fed8c2ce4 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -462,9 +462,9 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ }, }, { - name: "downshift", - description: "Switch to a fast/cheap model at the next action (works even without --downshift)", - acpDescription: "Downshift at the next action", + name: "prewalk", + description: "Switch to a fast/cheap model at the next action (works even without --prewalk)", + acpDescription: "Prewalk at the next action", handle: async (_command, runtime) => { const rolePattern = expandRoleAlias("pi/smol", runtime.settings); const resolved = resolveCliModel({ @@ -478,9 +478,9 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ if (!runtime.session.modelRegistry.hasConfiguredAuth(resolved.model)) { return usage(`No API key for ${resolved.model.provider}/${resolved.model.id}`, runtime); } - runtime.session.armDownshift(resolved.model, resolved.thinkingLevel); + runtime.session.armPrewalk(resolved.model, resolved.thinkingLevel); await runtime.output( - `Downshift on: switching to ${resolved.model.provider}/${resolved.model.id} at the next edit/write (todo-gated).`, + `Prewalk on: switching to ${resolved.model.provider}/${resolved.model.id} at the next edit/write (todo-gated).`, ); return commandConsumed(); }, diff --git a/packages/coding-agent/test/agent-session-downshift.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts similarity index 94% rename from packages/coding-agent/test/agent-session-downshift.test.ts rename to packages/coding-agent/test/agent-session-prewalk.test.ts index 9b697ba26..94a903870 100644 --- a/packages/coding-agent/test/agent-session-downshift.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -13,21 +13,21 @@ import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manage import { TempDir } from "@oh-my-pi/pi-utils"; /** - * Downshift: one-way switch from the starting model to a fast/cheap target + * Prewalk: one-way switch from the starting model to a fast/cheap target * at the first completed turn that starts execution — an edit/write tool, * or the todo-list init the plan nudge asks for — with a hidden plan nudge * before the switch and a hidden verify-before-finishing checklist after * it. This is the single mechanism that won out over fixed-turn and * ungated variants in benchmark testing — see the plan nudge / checklist / - * continuation-safety-net prompts under `src/prompts/system/downshift-*.md`. + * continuation-safety-net prompts under `src/prompts/system/prewalk-*.md`. */ -describe("AgentSession downshift", () => { +describe("AgentSession prewalk", () => { let tempDir: TempDir; let authStorage: AuthStorage; let session: AgentSession | undefined; beforeEach(async () => { - tempDir = TempDir.createSync("@pi-downshift-"); + tempDir = TempDir.createSync("@pi-prewalk-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db")); authStorage.setRuntimeApiKey("anthropic", "test-key"); }); @@ -110,7 +110,7 @@ describe("AgentSession downshift", () => { }); } - it("downshifts at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => { + it("prewalks at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); @@ -155,7 +155,7 @@ describe("AgentSession downshift", () => { settings: Settings.isolated({ "compaction.enabled": false }), modelRegistry, toolRegistry, - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -217,7 +217,7 @@ describe("AgentSession downshift", () => { settings: Settings.isolated({ "compaction.enabled": false }), modelRegistry, toolRegistry, - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -277,7 +277,7 @@ describe("AgentSession downshift", () => { [recordTool.name, recordTool as AgentTool], [writeTool.name, writeTool as AgentTool], ]), - downshift: { target }, + prewalk: { target }, }); await session.prompt("do the task"); @@ -292,13 +292,13 @@ describe("AgentSession downshift", () => { expect(session.model?.id).toBe(target.id); }); - it("armDownshift (the /downshift slash command) pre-arms the switch for the very next edit/write", async () => { + it("armPrewalk (the /prewalk slash command) pre-arms the switch for the very next edit/write", async () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - // No `downshift` in the session config — this simulates a session that - // was NOT started with --downshift, forced on via the slash command. + // No `prewalk` in the session config — this simulates a session that + // was NOT started with --prewalk, forced on via the slash command. const mock = createMockModel({ responses: [toolCall("t1", "write"), { content: ["done"] }] }); const requested: string[] = []; const agent = new Agent({ @@ -325,8 +325,8 @@ describe("AgentSession downshift", () => { }); // Arming twice back-to-back must stay a single, idempotent arm. - session.armDownshift(target); - session.armDownshift(target); + session.armPrewalk(target); + session.armPrewalk(target); await session.prompt("do the task"); diff --git a/packages/harbor-manager/README.md b/packages/harbor-manager/README.md index f83b3c54d..22bb1bc4f 100644 --- a/packages/harbor-manager/README.md +++ b/packages/harbor-manager/README.md @@ -48,7 +48,7 @@ bun run serve --port 4700 ``` `benchmark` is `harbor`, `edit`, or `snapcompact`. Harbor uses `dataset`, - `include`, `timeoutMultiplier`, and `downshift`; edit uses `include` as task IDs; + `include`, `timeoutMultiplier`, and `prewalk`; edit uses `include` as task IDs; SnapCompact uses `conditions` and treats `tasks` as the passage limit. - `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read). - `DELETE /api/runs/:name` — cancel a manager-launched run. diff --git a/packages/harbor-manager/src/experiments.test.ts b/packages/harbor-manager/src/experiments.test.ts index cd0e8f630..053f068cc 100644 --- a/packages/harbor-manager/src/experiments.test.ts +++ b/packages/harbor-manager/src/experiments.test.ts @@ -24,7 +24,7 @@ function runRow(overrides: Partial): RunRow { agent: "omp", models: "anthropic/claude-opus-4-8", label: "", - downshift: null, + prewalk: null, config: {}, role: "", note: "", @@ -125,11 +125,11 @@ describe("summarizeArm", () => { expect(finished.costPerTask).toBeCloseTo(1, 5); }); - it("describes the downshift config in the arm line", () => { + it("describes the prewalk config in the arm line", () => { const arm = summarizeArm( runRow({ jobName: "sb2-nact", - downshift: JSON.stringify({ into: "google/gemini-3.5-flash" }), + prewalk: JSON.stringify({ into: "google/gemini-3.5-flash" }), }), [], ); @@ -140,7 +140,7 @@ describe("summarizeArm", () => { const arm = summarizeArm( runRow({ jobName: "sb2-nact", - downshift: JSON.stringify({ model: "google/gemini-3.5-flash", onAction: true, plan: true }), + prewalk: JSON.stringify({ model: "google/gemini-3.5-flash", onAction: true, plan: true }), }), [], ); diff --git a/packages/harbor-manager/src/experiments.ts b/packages/harbor-manager/src/experiments.ts index 9f19c97ef..423084c1f 100644 --- a/packages/harbor-manager/src/experiments.ts +++ b/packages/harbor-manager/src/experiments.ts @@ -19,7 +19,7 @@ export interface ArmSummary { run: RunRow; /** Arm label: job name minus the experiment prefix. */ arm: string; - /** Human config line: models plus downshift description when known. */ + /** Human config line: models plus prewalk description when known. */ config: string; /** Observed pass% over decided trials. */ passPct: number | null; @@ -67,11 +67,11 @@ export function armOf(jobName: string): string { return jobName.length > exp.length ? jobName.slice(exp.length + 1) : jobName; } -function downshiftLabel(downshiftJson: string | null): string { - if (!downshiftJson) return ""; +function prewalkLabel(prewalkJson: string | null): string { + if (!prewalkJson) return ""; try { // Historical rows may hold legacy reasoning-slide JSON ({model, turns, onAction, plan}). - const parsed = JSON.parse(downshiftJson) as { + const parsed = JSON.parse(prewalkJson) as { into?: string; model?: string; turns?: number; @@ -118,7 +118,7 @@ export function summarizeArm(run: RunRow, traces: TraceRow[]): ArmSummary { return { run, arm: armOf(run.jobName), - config: `${run.benchmark} · ${run.models}${downshiftLabel(run.downshift)}`, + config: `${run.benchmark} · ${run.models}${prewalkLabel(run.prewalk)}`, passPct, costPerTask, meanTrialMs, diff --git a/packages/harbor-manager/src/manager.test.ts b/packages/harbor-manager/src/manager.test.ts index 42e6f3224..5869f8ed3 100644 --- a/packages/harbor-manager/src/manager.test.ts +++ b/packages/harbor-manager/src/manager.test.ts @@ -145,7 +145,7 @@ describe("RunStore", () => { store.discover(); store.setExperimentGoal("exp", "does the treatment beat the baseline?"); expect(store.setRunMeta("exp-base", { role: "baseline", note: "plain model" })).toBe(true); - expect(store.setRunMeta("exp-treat", { role: "variant", note: "downshift flash", label: "flash@edit" })).toBe( + expect(store.setRunMeta("exp-treat", { role: "variant", note: "prewalk flash", label: "flash@edit" })).toBe( true, ); expect(store.setRunMeta("exp-missing", { role: "variant" })).toBe(false); @@ -155,15 +155,15 @@ describe("RunStore", () => { // ArmSummary.arm resolves to the display label when one is set. expect(detail?.arms.map(a => [a.arm, a.run.role, a.run.note, a.run.label])).toEqual([ ["base", "baseline", "plain model", ""], - ["flash@edit", "variant", "downshift flash", "flash@edit"], + ["flash@edit", "variant", "prewalk flash", "flash@edit"], ]); // Partial updates keep the omitted fields. - expect(store.setRunMeta("exp-treat", { note: "downshift flash v2" })).toBe(true); + expect(store.setRunMeta("exp-treat", { note: "prewalk flash v2" })).toBe(true); const treat = store.getRun("exp-treat"); expect(treat?.label).toBe("flash@edit"); expect(treat?.role).toBe("variant"); - expect(treat?.note).toBe("downshift flash v2"); + expect(treat?.note).toBe("prewalk flash v2"); }); it("finalizes running rows whose owning process died", () => { @@ -328,8 +328,8 @@ describe("resolveArmLaunch", () => { arm: "n8", model: "google/gemini-3.5-flash", role: "variant", - note: "downshift@flash", - downshift: { into: "google/gemini-3.5-flash" }, + note: "prewalk@flash", + prewalk: { into: "google/gemini-3.5-flash" }, }); expect(launch.jobName).toBe("exp-n8"); @@ -340,7 +340,7 @@ describe("resolveArmLaunch", () => { expect(launch.timeoutMultiplier).toBe(2); expect(launch.model).toBe("google/gemini-3.5-flash"); expect(launch.role).toBe("variant"); - expect(launch.downshift?.into).toBe("google/gemini-3.5-flash"); + expect(launch.prewalk?.into).toBe("google/gemini-3.5-flash"); }); it("prefers the sibling with a recorded include list over newer include-less siblings", () => { diff --git a/packages/harbor-manager/src/runner.test.ts b/packages/harbor-manager/src/runner.test.ts index 922e74de0..8723771d3 100644 --- a/packages/harbor-manager/src/runner.test.ts +++ b/packages/harbor-manager/src/runner.test.ts @@ -7,13 +7,13 @@ describe("generic agent-arg / env passthrough", () => { "--model", "anthropic/claude-opus-4-8", "--agent-arg", - "--downshift", + "--prewalk", "--agent-arg", - "--downshift-into", + "--prewalk-into", "--agent-arg", "google/gemini-3.5-flash", ]); - expect(cfg.agentArgs).toEqual(["--downshift", "--downshift-into", "google/gemini-3.5-flash"]); + expect(cfg.agentArgs).toEqual(["--prewalk", "--prewalk-into", "google/gemini-3.5-flash"]); const env = buildHarborEnv(cfg, "/tmp/models.yml", null, "test"); expect(JSON.parse(env.OMP_BENCH_AGENT_ARGS ?? "[]")).toEqual(cfg.agentArgs); diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index edf24119f..d7526481d 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -51,8 +51,8 @@ export interface LaunchRequest { agent?: string; jobName?: string; webSearch?: boolean; - /** Downshift to a fast/cheap model at the first edit/write once the todo list exists; `into` overrides the default "smol" target. */ - downshift?: { into?: string }; + /** Prewalk to a fast/cheap model at the first edit/write once the todo list exists; `into` overrides the default "smol" target. */ + prewalk?: { into?: string }; /** Role of this run inside its experiment (baseline vs treatment). */ role?: RunRole; /** One-line description of what this arm tests. */ @@ -70,7 +70,7 @@ export interface AddArmRequest { /** Arm label; becomes the `-` job name. */ arm: string; model: string; - downshift?: LaunchRequest["downshift"]; + prewalk?: LaunchRequest["prewalk"]; /** Explicit task sample; skips sibling inheritance when provided. */ include?: string[]; role?: RunRole; @@ -110,7 +110,7 @@ function parseServerArgs(argv: string[]): { port: number; jobsDir: string } { * Inherits the experiment's benchmark, dataset, and — crucially — the exact * task sample from a sibling arm (its recorded `include`, else its observed * trial tasks) so the arm is directly comparable. Only per-arm knobs (model, - * downshift, role, note, extra args) come from `req`. Throws if the experiment has + * prewalk, role, note, extra args) come from `req`. Throws if the experiment has * no runs to inherit from or the arm name is taken. */ export function resolveArmLaunch(store: RunStore, experimentId: string, req: AddArmRequest): LaunchRequest { @@ -174,7 +174,7 @@ export function resolveArmLaunch(store: RunStore, experimentId: string, req: Add prebuiltBinaries: cfg.prebuiltBinaries === true || undefined, conditions: conditions.length > 0 ? conditions : undefined, jobName, - downshift: req.downshift, + prewalk: req.prewalk, role: req.role, note: req.note, extraArgs: req.extraArgs, @@ -392,12 +392,12 @@ export class ManagerServer { argv.push("--timeout-multiplier", String(request.timeoutMultiplier)); if (request.webSearch) argv.push("--web-search"); for (const task of request.include ?? []) argv.push("--include", task); - if (request.downshift) { - argv.push("--agent-arg", "--downshift"); - if (request.downshift.into) { - argv.push("--agent-arg", "--downshift-into", "--agent-arg", request.downshift.into); - const provider = request.downshift.into.split("/", 1)[0]; - if (provider && request.downshift.into.includes("/")) argv.push("--providers", provider); + if (request.prewalk) { + argv.push("--agent-arg", "--prewalk"); + if (request.prewalk.into) { + argv.push("--agent-arg", "--prewalk-into", "--agent-arg", request.prewalk.into); + const provider = request.prewalk.into.split("/", 1)[0]; + if (provider && request.prewalk.into.includes("/")) argv.push("--providers", provider); } } if (request.prebuiltBinaries) { @@ -434,7 +434,7 @@ export class ManagerServer { dataset, agent: request.agent ?? "omp", models: [request.model], - downshift: request.downshift, + prewalk: request.prewalk, config: { ...request }, pid: proc.pid, role: request.role, diff --git a/packages/harbor-manager/src/store.ts b/packages/harbor-manager/src/store.ts index 960338357..5a7c193bc 100644 --- a/packages/harbor-manager/src/store.ts +++ b/packages/harbor-manager/src/store.ts @@ -27,13 +27,13 @@ export interface RunRow { dataset: string; agent: string; models: string; - /** JSON downshift config (`{ into?: string }`); older rows may hold legacy reasoning-slide JSON. */ - downshift: string | null; + /** JSON prewalk config (`{ into?: string }`); older rows may hold legacy reasoning-slide JSON. */ + prewalk: string | null; /** Benchmark-specific launch configuration. */ config: Record; /** Role inside the experiment (baseline vs treatment); "" when unspecified. */ role: RunRole; - /** One-line description of what this arm tests (e.g. "downshift→flash at first edit/write"). */ + /** One-line description of what this arm tests (e.g. "prewalk→flash at first edit/write"). */ note: string; /** Display-name override for the arm; "" falls back to the jobName-derived arm label. */ label: string; @@ -78,7 +78,7 @@ export interface LaunchRecord { dataset: string; agent: string; models: string[]; - downshift?: { into?: string }; + prewalk?: { into?: string }; pid: number; role?: RunRole; note?: string; @@ -92,7 +92,7 @@ CREATE TABLE IF NOT EXISTS runs ( dataset TEXT NOT NULL DEFAULT '', agent TEXT NOT NULL DEFAULT 'omp', models TEXT NOT NULL DEFAULT '', - downshift TEXT, + prewalk TEXT, role TEXT NOT NULL DEFAULT '', note TEXT NOT NULL DEFAULT '', label TEXT NOT NULL DEFAULT '', @@ -200,11 +200,11 @@ export class RunStore { if (!runColumns.has("metrics_json")) { this.#db.run("ALTER TABLE runs ADD COLUMN metrics_json TEXT NOT NULL DEFAULT '{}'"); } - if (runColumns.has("slide") && !runColumns.has("downshift")) { - this.#db.run("ALTER TABLE runs RENAME COLUMN slide TO downshift"); + if (runColumns.has("slide") && !runColumns.has("prewalk")) { + this.#db.run("ALTER TABLE runs RENAME COLUMN slide TO prewalk"); } - if (!runColumns.has("slide") && !runColumns.has("downshift")) { - this.#db.run("ALTER TABLE runs ADD COLUMN downshift TEXT"); + if (!runColumns.has("slide") && !runColumns.has("prewalk")) { + this.#db.run("ALTER TABLE runs ADD COLUMN prewalk TEXT"); } const traceColumns = new Set( (this.#db.query("PRAGMA table_info(trials)").all() as Array<{ name: string }>).map(c => c.name), @@ -222,7 +222,7 @@ export class RunStore { this.#db .query( `INSERT INTO runs - (job_name, benchmark, dataset, agent, models, downshift, role, note, config_json, status, pid, created_at) + (job_name, benchmark, dataset, agent, models, prewalk, role, note, config_json, status, pid, created_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'running', ?, ?) ON CONFLICT(job_name) DO UPDATE SET benchmark = excluded.benchmark, pid = excluded.pid, status = 'running', @@ -236,7 +236,7 @@ export class RunStore { launch.dataset, launch.agent, launch.models.join(","), - launch.downshift ? JSON.stringify(launch.downshift) : null, + launch.prewalk ? JSON.stringify(launch.prewalk) : null, launch.role ?? "", launch.note ?? "", JSON.stringify(launch.config ?? {}), @@ -459,7 +459,7 @@ function rowToRun(r: Record): RunRow { dataset: String(r.dataset), agent: String(r.agent), models: String(r.models), - downshift: r.downshift === null ? null : String(r.downshift), + prewalk: r.prewalk === null ? null : String(r.prewalk), config: JSON.parse(String(r.config_json ?? "{}")), role: String(r.role ?? "") as RunRole, note: String(r.note ?? ""), diff --git a/packages/harbor-manager/src/web/app.tsx b/packages/harbor-manager/src/web/app.tsx index d951f4719..5d0936b79 100644 --- a/packages/harbor-manager/src/web/app.tsx +++ b/packages/harbor-manager/src/web/app.tsx @@ -22,7 +22,7 @@ interface RunRow { dataset: string; agent: string; models: string; - downshift: string | null; + prewalk: string | null; config: Record; role: RunRole; note: string; @@ -747,7 +747,7 @@ const CELL_CLASS: Record = { /** * The comparison anchor for an experiment: the completed baseline arm with the - * highest pass rate (the "ceiling" a downshift arm tries to preserve). Ties + * highest pass rate (the "ceiling" a prewalk arm tries to preserve). Ties * break toward the cheaper arm. Returns null when no baseline has finished data. */ function pickReferenceArm(arms: ArmSummary[]): ArmSummary | null { @@ -799,7 +799,7 @@ function Delta({ /** * Launch a new arm into an existing experiment. The server inherits the * experiment's dataset and exact task sample from a sibling arm, so only the - * arm-specific knobs (name, model, role, note, optional downshift) are collected here. + * arm-specific knobs (name, model, role, note, optional prewalk) are collected here. */ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () => void }) { const [msg, setMsg] = useState(""); @@ -810,8 +810,8 @@ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () const body: Record = { arm: f.get("arm"), model: f.get("model") }; if (f.get("role")) body.role = f.get("role"); if (f.get("note")) body.note = f.get("note"); - if (f.get("downshiftInto") || f.get("downshift")) { - body.downshift = f.get("downshiftInto") ? { into: f.get("downshiftInto") } : {}; + if (f.get("prewalkInto") || f.get("prewalk")) { + body.prewalk = f.get("prewalkInto") ? { into: f.get("prewalkInto") } : {}; } setMsg("launching…"); const res = await fetch(`/api/experiments/${encodeURIComponent(experimentId)}/arms`, { @@ -838,9 +838,9 @@ function AddArmForm({ experimentId, onDone }: { experimentId: string; onDone: () - +
+ ) : ( + r.benchmark === "harbor" && + (r.done < r.nTotal || r.error > 0) && ( + + ) )} From a886a309018ef043ffd8565c2a376cc084e42885 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 16:23:00 +0200 Subject: [PATCH 06/28] feat(hashline): standardized drift recovery using anchor remapping - Replaced 3-way-merge and session-chain replay strategies with a consistent anchor remapping flow for drift recovery. - Removed legacy reconciliation logic and associated session-replay warning constants. - Updated recovery flow to mandate anchor consistency and validate against duplicated context. - Refined test suite to verify anchor mapping mechanics and explicit refusal of ambiguous remappings. --- packages/hashline/CHANGELOG.md | 1 + packages/hashline/src/messages.ts | 9 - packages/hashline/src/patcher.ts | 8 +- packages/hashline/src/recovery.ts | 185 +++--------------- packages/hashline/src/snapshots.ts | 7 +- packages/hashline/test/block.test.ts | 2 +- .../hashline/test/boundary-repair.test.ts | 6 +- .../test/recovery-session-chain.test.ts | 48 +++-- 8 files changed, 73 insertions(+), 193 deletions(-) diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index bd0e21a57..897f05038 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -6,6 +6,7 @@ - Rejected ambiguous swaps that risk silent deletion of range boundaries - Prevented ambiguous auto-repairing of structural closing lines when payload placement is unclear +- Prevented stale-hash recovery from relocating edits onto duplicated context after the original target changed ## [16.3.3] - 2026-07-02 diff --git a/packages/hashline/src/messages.ts b/packages/hashline/src/messages.ts index 5c93c70e7..458585143 100644 --- a/packages/hashline/src/messages.ts +++ b/packages/hashline/src/messages.ts @@ -208,15 +208,6 @@ export const RECOVERY_EXTERNAL_WARNING = export const RECOVERY_SESSION_CHAIN_WARNING = "Recovered from a stale file hash using an earlier in-session snapshot (a prior edit in this session advanced the hash)."; -/** - * `Recovery`: session-chain replay fast-path. Less certain than - * {@link RECOVERY_SESSION_CHAIN_WARNING} — the 3-way merge refused, the - * anchor-content gate passed, but a coincidental insert+delete earlier in - * the chain could still misplace an anchor — hence the verify hedge. - */ -export const RECOVERY_SESSION_REPLAY_WARNING = - "Recovered by replaying your edits onto the current file content (a prior in-session edit changed the lines you re-targeted with a stale hash). Verify the diff matches your intent."; - /** `Recovery`: stale anchors were relocated to unchanged live lines after drift. */ export const RECOVERY_LINE_REMAP_WARNING = "Recovered by remapping stale line anchors to unchanged current lines (file changed since the tagged read). Verify the diff matches your intent."; diff --git a/packages/hashline/src/patcher.ts b/packages/hashline/src/patcher.ts index d4d98488d..b49156135 100644 --- a/packages/hashline/src/patcher.ts +++ b/packages/hashline/src/patcher.ts @@ -586,7 +586,7 @@ export class Patcher { const expected = exists ? section.fileHash : undefined; // The 4-hex tag is content-derived: when the live text hashes to it, // trust the match and apply directly. `storedSnapshotForTag` feeds the - // drift paths below (block resolution, 3-way recovery); on a 16-bit + // drift paths below (block resolution, anchor remapping); on a 16-bit // tag collision it resolves to the most-recently recorded text. const storedSnapshotForTag = expected === undefined ? null : this.snapshots.byHash(canonicalPath, expected); const liveMatches = expected !== undefined && computeFileHash(normalized) === expected; @@ -598,7 +598,7 @@ export class Patcher { // - live content matches the tag (or there is no tag) → resolve against // the live, normalized content; // - the file drifted → resolve against the tagged snapshot's text so the - // resulting ranges flow through the 3-way-merge recovery below. + // resulting ranges can be mapped to unchanged live lines below. // When a block edit needs the tagged snapshot but it is unavailable, the // range cannot be placed safely — reject with a MismatchError (re-read). const blockResolutions: BlockResolution[] = []; @@ -640,8 +640,8 @@ export class Patcher { const result = applyEdits(normalized, resolved); return withResolveWarnings({ ...result, warnings: [HEADTAIL_DRIFT_WARNING, ...(result.warnings ?? [])] }); } - // File drifted: try to replay the edit against the version the tag - // names and 3-way-merge it onto the live content. + // File drifted: map every anchor from the tagged snapshot to unchanged + // live lines. Recovery refuses changed or ambiguous targets. const recovered = this.recovery.tryRecover({ path: canonicalPath, currentText: normalized, diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts index b3e1b9ff4..eb48cee9e 100644 --- a/packages/hashline/src/recovery.ts +++ b/packages/hashline/src/recovery.ts @@ -1,29 +1,17 @@ /** - * Recover from a stale section snapshot tag by replaying the would-be edit - * against a cached pre-edit snapshot of the file and 3-way-merging the - * result onto the current on-disk content. + * Recovers stale section tags by proving that every anchored line still maps + * to one unchanged, contiguous region in the current file, then replaying the + * edit against that live content. * - * The patcher consults this when a section tag resolves to a snapshot that no - * longer matches the live file content. The recovery class is stateless apart - * from the {@link SnapshotStore} it queries; the snapshot store is the seam - * lets you plug in your own caching strategy. + * Recovery fails closed when the target changed or became ambiguous. The + * patcher then returns a mismatch with fresh context instead of guessing. */ import * as Diff from "diff"; import { applyEdits } from "./apply"; -import { - RECOVERY_EXTERNAL_WARNING, - RECOVERY_LINE_REMAP_WARNING, - RECOVERY_SESSION_CHAIN_WARNING, - RECOVERY_SESSION_REPLAY_WARNING, -} from "./messages"; -import type { Snapshot, SnapshotStore } from "./snapshots"; +import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages"; +import type { SnapshotStore } from "./snapshots"; import type { Anchor, ApplyResult, Edit } from "./types"; -// Section tags are line-precise; never let Diff.applyPatch slide a hunk -// onto a duplicate closer 100+ lines away. If snapshot replay does not -// align exactly, refuse and let the caller re-read. -const RECOVERY_FUZZ_FACTOR = 0; - export interface RecoveryArgs { path: string; currentText: string; @@ -40,31 +28,6 @@ export interface RecoveryResult { warnings: string[]; } -function applyEditsToSnapshot( - previousText: string, - currentText: string, - edits: readonly Edit[], - recoveryWarning: string, -): RecoveryResult | null { - let applied: ApplyResult; - try { - applied = applyEdits(previousText, [...edits]); - } catch { - return null; - } - if (applied.text === previousText) return null; - - const patch = Diff.structuredPatch("file", "file", previousText, applied.text, "", "", { context: 3 }); - const merged = Diff.applyPatch(currentText, patch, { fuzzFactor: RECOVERY_FUZZ_FACTOR }); - if (typeof merged !== "string" || merged === currentText) return null; - - const firstChangedLine = findFirstChangedLine(currentText, merged) ?? applied.firstChangedLine; - const hasNetChange = firstChangedLine !== undefined; - const warnings = hasNetChange ? [recoveryWarning, ...(applied.warnings ?? [])] : [...(applied.warnings ?? [])]; - - return { text: merged, firstChangedLine, warnings }; -} - function collectAnchorLines(edits: readonly Edit[]): number[] { const lines: number[] = []; for (const edit of edits) { @@ -81,27 +44,6 @@ function getEditAnchors(edit: Edit): Anchor[] { return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : []; } -/** - * Returns true when every anchor line in `edits` has identical content in - * `previousText` and `currentText`. The session-chain replay fast-path - * requires this: if the prior in-session edit rewrote the line the model is - * now re-targeting with a stale hash, replaying onto current would silently - * overwrite the new content with whatever the model authored against the - * old content — a corruption window, not a recovery. - */ -function verifyAnchorContent(previousText: string, currentText: string, edits: readonly Edit[]): boolean { - const lines = collectAnchorLines(edits); - if (lines.length === 0) return true; - const prev = previousText.split("\n"); - const curr = currentText.split("\n"); - for (const line of lines) { - const idx = line - 1; - if (idx < 0 || idx >= prev.length || idx >= curr.length) return false; - if (prev[idx] !== curr[idx]) return false; - } - return true; -} - function buildLineMap(previousText: string, currentText: string): Map { const previousLines = previousText.split("\n"); const currentLines = currentText.split("\n"); @@ -198,7 +140,7 @@ function validateUniqueAnchorContext( ): boolean { const offset = mapped - line; const { before, after } = neighbors; - if (after !== undefined) return lineMap.get(after) === after + offset; + if (after !== undefined && lineMap.get(after) === after + offset) return true; return before !== undefined && lineMap.get(before) === before + offset; } @@ -237,7 +179,12 @@ function validateRemappedAnchorContext( return true; } -function remapEditsToCurrent(previousText: string, currentText: string, edits: readonly Edit[]): Edit[] | null { +interface RemappedEdits { + edits: Edit[]; + offset: number; +} + +function remapEditsToCurrent(previousText: string, currentText: string, edits: readonly Edit[]): RemappedEdits | null { const lineMap = buildLineMap(previousText, currentText); if (!validateRemappedAnchorContext(previousText, currentText, lineMap, edits)) return null; const offsets: number[] = []; @@ -289,21 +236,21 @@ function remapEditsToCurrent(previousText: string, currentText: string, edits: r if (offsets.length === 0) return null; const firstOffset = offsets[0]; - if (firstOffset === 0) return null; if (!offsets.every(offset => offset === firstOffset)) return null; - return remapped; + return { edits: remapped, offset: firstOffset }; } function replayRemappedAnchorsOnCurrent( previousText: string, currentText: string, edits: readonly Edit[], + recoveryWarning: string, ): RecoveryResult | null { const remapped = remapEditsToCurrent(previousText, currentText, edits); if (remapped === null) return null; let applied: ApplyResult; try { - applied = applyEdits(currentText, remapped); + applied = applyEdits(currentText, remapped.edits); } catch { return null; } @@ -311,78 +258,18 @@ function replayRemappedAnchorsOnCurrent( return { text: applied.text, firstChangedLine: applied.firstChangedLine, - warnings: [RECOVERY_LINE_REMAP_WARNING, ...(applied.warnings ?? [])], + warnings: [remapped.offset === 0 ? recoveryWarning : RECOVERY_LINE_REMAP_WARNING, ...(applied.warnings ?? [])], }; } - -function replaySessionChainOnCurrent( - previousText: string, - currentText: string, - edits: readonly Edit[], -): RecoveryResult | null { - // Two guards narrow the corruption window. Neither alone is sufficient, - // and even together they don't fully prove correctness — replay is the - // less-certain recovery mode and emits RECOVERY_SESSION_REPLAY_WARNING - // so the caller can verify the diff. - // - Equal line counts: every line number in `edits` still resolves to - // SOME logical row (no net shift across the prior chain). A - // coincidental insert+delete pair can still leave indices pointing - // at different logical rows than the model anchored against. - // - Anchor-content alignment: the row at each anchor's line index has - // identical content in previous and current. Catches the common - // case of a prior edit rewriting the targeted line; can still be - // coincidentally satisfied by a duplicated row at the shifted - // index. - if (previousText.split("\n").length !== currentText.split("\n").length) return null; - if (!verifyAnchorContent(previousText, currentText, edits)) return null; - let applied: ApplyResult; - try { - applied = applyEdits(currentText, [...edits]); - } catch { - return null; - } - if (applied.text === currentText) return null; - return { - text: applied.text, - firstChangedLine: applied.firstChangedLine, - warnings: [RECOVERY_SESSION_REPLAY_WARNING, ...(applied.warnings ?? [])], - }; -} - -/** First 1-indexed line at which `a` and `b` diverge, or `undefined` if equal. */ -function findFirstChangedLine(a: string, b: string): number | undefined { - if (a === b) return undefined; - const aLines = a.split("\n"); - const bLines = b.split("\n"); - const max = Math.max(aLines.length, bLines.length); - for (let i = 0; i < max; i++) { - if (aLines[i] !== bLines[i]) return i + 1; - } - return undefined; -} - -function isHeadSnapshot(head: Snapshot | null, snapshot: Snapshot): boolean { - return head === snapshot; -} - /** * Stateless recovery driver over a {@link SnapshotStore}. Construct once and - * call {@link Recovery.tryRecover} per stale-tag incident. The default - * implementation tries three strategies in order: + * call {@link Recovery.tryRecover} per stale-tag incident. * - * 1. Apply the edits on the full-file version the tag names, then 3-way-merge - * the resulting patch onto the live content (handles external writes). - * 2. Remap every stale anchor through the unchanged-line diff from the tagged - * snapshot to the live text, then replay on live content. This handles a - * prior insertion/deletion before the target while refusing changed anchors - * and mixed offsets across the same edit range. - * 3. (Session chain) If that version wasn't the head, replay the edits onto - * the live content directly when line counts match AND every edit's anchor - * line content is unchanged between version and current — a prior in-session - * edit advanced the tag and the model's anchors still name the same logical - * rows. Emits a dedicated {@link RECOVERY_SESSION_REPLAY_WARNING} because - * even with both guards a coincidental insert+delete pair on duplicate rows - * can still land the edit on the wrong row; see {@link replaySessionChainOnCurrent}. + * Recovery maps every stale anchor through unchanged lines from the tagged + * snapshot to the live text, validates surrounding context, and replays the + * edit directly on live content. All anchors must move by one consistent + * offset. A changed, deleted, split, or ambiguous target is rejected so the + * caller can surface a {@link MismatchError} with current context. */ export class Recovery { constructor(readonly store: SnapshotStore) {} @@ -392,26 +279,12 @@ export class Recovery { */ tryRecover(args: RecoveryArgs): RecoveryResult | null { const { path, currentText, fileHash, edits } = args; - // When two retained texts collide on the 16-bit tag, resolve to the - // most-recently recorded one; a wrong pick can only land if one of the - // merge/remap/session-chain strategies below applies it cleanly. + // When retained texts collide on the 16-bit tag, use the latest one. + // Recovery still requires its anchors and context to map unambiguously. const snapshot = this.store.byHash(path, fileHash); if (!snapshot) return null; - const isHead = isHeadSnapshot(this.store.head(path), snapshot); - const recoveryWarning = isHead ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; - const merged = applyEditsToSnapshot(snapshot.text, currentText, edits, recoveryWarning); - if (merged !== null) return merged; - // Line-shift fallback: the 3-way merge refused, but unchanged anchor - // lines may have moved because a prior edit inserted or deleted rows - // before them. Remap only when every anchor resolves through the diff - // with one consistent offset; otherwise the edit range was touched. - const remapped = replayRemappedAnchorsOnCurrent(snapshot.text, currentText, edits); - if (remapped !== null) return remapped; - // Session-chain fallback: replay onto current is gated by line-count - // equality AND anchor-content alignment — see - // `replaySessionChainOnCurrent` for why both guards together still - // don't fully prove correctness. - if (!isHead) return replaySessionChainOnCurrent(snapshot.text, currentText, edits); - return null; + const recoveryWarning = + this.store.head(path) === snapshot ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; + return replayRemappedAnchorsOnCurrent(snapshot.text, currentText, edits, recoveryWarning); } } diff --git a/packages/hashline/src/snapshots.ts b/packages/hashline/src/snapshots.ts index 97189ec3b..874433b2b 100644 --- a/packages/hashline/src/snapshots.ts +++ b/packages/hashline/src/snapshots.ts @@ -11,8 +11,7 @@ * {@link SnapshotStore.record} with the full normalized text they observed. * The store hashes it, dedups against the per-path history, and returns the * tag. Consumers (recovery, the patcher) resolve a stale tag back to the - * recorded full text via {@link SnapshotStore.byHash} and 3-way-merge the - * would-be edit onto the live content. + * recorded full text and map its unchanged edit anchors onto live content. * * The abstract base class lets callers plug in whatever storage they like * (LRU, persistent SQLite, etc.). {@link InMemorySnapshotStore} ships as a @@ -199,8 +198,8 @@ export class InMemorySnapshotStore extends SnapshotStore { // texts that happen to share the 4-hex tag are DIFFERENT snapshots — fusing // them under one entry would corrupt seenLines (attaching lines from // text B onto the stored text A) and let the patcher misresolve which - // snapshot the section tag names when it does 3-way merge or seen-line - // validation. See issue #4075. + // snapshot the section tag names during recovery or seen-line validation. + // See issue #4075. const existing = history.find(version => version.hash === hash && version.text === fullText); if (existing) { // Same content state observed again: refresh recency and promote to diff --git a/packages/hashline/test/block.test.ts b/packages/hashline/test/block.test.ts index d6eb03b32..2526198b7 100644 --- a/packages/hashline/test/block.test.ts +++ b/packages/hashline/test/block.test.ts @@ -229,7 +229,7 @@ describe("Patcher with a block resolver", () => { const patcher = new Patcher({ fs, snapshots, blockResolver: stubResolver }); // `block 2` resolves against the SNAPSHOT → span [2,3] → replace - // "line1","line2"; recovery 3-way-merges the change onto the live file. + // "line1","line2"; recovery maps that unchanged span onto the live file. const result = await patcher.apply(Patch.parse(`[${PATH}#${tag}]\nSWAP.BLK 2:\n+NEW`)); expect(result.sections[0]?.op).toBe("update"); diff --git a/packages/hashline/test/boundary-repair.test.ts b/packages/hashline/test/boundary-repair.test.ts index cdf90aa35..6c53e9a06 100644 --- a/packages/hashline/test/boundary-repair.test.ts +++ b/packages/hashline/test/boundary-repair.test.ts @@ -607,8 +607,8 @@ describe("boundary-balance repair through stale-snapshot recovery", () => { // Recovery composes `applyEdits` to compute the intended change, so the // boundary repair runs there too. The snapshot (what the model read) // carries the structure; the live file has drifted far from the edit - // region, so the stale-hash 3-way merge succeeds and the repaired - // (de-duplicated) hunk lands without doubling the closer. + // region, so anchor recovery succeeds and the repaired (de-duplicated) + // hunk lands without doubling the closer. it("de-duplicates a closer while recovering from a drifted file", () => { const snapshotLines = [ 'import { x } from "y";', @@ -628,7 +628,7 @@ describe("boundary-balance repair through stale-snapshot recovery", () => { ]; const snapshotText = `${snapshotLines.join("\n")}\n`; // Live file drifted only at the tail (line 13) — far outside the edit - // region (lines 4-6), so the 3-way merge applies cleanly. + // region (lines 4-6), so unchanged-anchor recovery succeeds. const currentText = snapshotText.replace("const tail = 0;", "const tail = 99;"); const store = new InMemorySnapshotStore(); diff --git a/packages/hashline/test/recovery-session-chain.test.ts b/packages/hashline/test/recovery-session-chain.test.ts index 08db6634e..1fbbb85e9 100644 --- a/packages/hashline/test/recovery-session-chain.test.ts +++ b/packages/hashline/test/recovery-session-chain.test.ts @@ -15,7 +15,7 @@ import { InMemorySnapshotStore, parsePatch, RECOVERY_LINE_REMAP_WARNING, - RECOVERY_SESSION_REPLAY_WARNING, + RECOVERY_SESSION_CHAIN_WARNING, Recovery, } from "@oh-my-pi/hashline"; @@ -58,10 +58,9 @@ describe("Recovery — session-chain replay anchor-content gate", () => { it("replays edits onto current when every anchor's line content is unchanged", () => { const { store, v1Text, h0 } = seedTwoSnapshots(); - // Edit anchored at line 3 — unchanged between v0 and v1. The 3-way - // merge fails (patch context includes the rewritten line 5), but the - // replay fallback is safe because the model's anchor still names the - // same logical content. + // Edit anchored at line 3 — unchanged between v0 and v1. Recovery + // proves that the target and its surrounding context still map to the + // same live lines before replaying the edit. const { edits } = parsePatch("SWAP 3.=3:\n|L3-MODEL"); const recovered = new Recovery(store).tryRecover({ @@ -76,11 +75,10 @@ describe("Recovery — session-chain replay anchor-content gate", () => { // Prior in-session change must survive — the model's edit lands on // top of current, not on top of the stale snapshot. expect(recovered?.text).toContain("L5-CHANGED"); - // The replay path is the less-certain recovery mode (a coincidental - // insert+delete pair earlier in the chain could leave indices - // pointing at duplicated rows even with both guards satisfied), so - // the dedicated REPLAY warning surfaces a "verify the diff" hedge. - expect(recovered?.warnings).toContain(RECOVERY_SESSION_REPLAY_WARNING); + // Zero-offset recovery against an earlier retained snapshot reports the + // session-chain banner; unlike the removed direct replay fallback, this + // path has proved the anchors through the unchanged-line map. + expect(recovered?.warnings).toContain(RECOVERY_SESSION_CHAIN_WARNING); }); it("recovers stale anchors shifted by a prior in-session insertion", () => { @@ -141,11 +139,30 @@ describe("Recovery — session-chain replay anchor-content gate", () => { expect(recovered).toBeNull(); }); - it("refuses unique-line remaps when following context no longer matches", () => { + it("refuses to relocate a stale replacement onto duplicated context", () => { + const store = new InMemorySnapshotStore(); + const block = ["head", "TARGET_A", "TARGET_B", "ctx1", "ctx2", "ctx3"]; + const v0Text = lines(...block, "middle", ...block, "tail"); + const hash = store.record(PATH, v0Text); + const currentText = lines("head", "CHANGED_A", "CHANGED_B", "ctx1", "ctx2", "ctx3", "middle", ...block, "tail"); + const { edits } = parsePatch("SWAP 2.=3:\n+MODEL_A\n+MODEL_B"); + + const recovered = new Recovery(store).tryRecover({ + path: PATH, + currentText, + fileHash: hash, + edits, + }); + + expect(recovered).toBeNull(); + expect(currentText).toContain("TARGET_A\nTARGET_B"); + }); + + it("refuses an isolated unique-line remap when neither neighbor follows its offset", () => { const store = new InMemorySnapshotStore(); const v0Text = lines("L1", "L2", "L3", "L4", "T", "L6"); const h0 = store.record(PATH, v0Text); - const v1Text = lines("X", "L1", "L2", "L3", "L4", "T", "T_CHANGED", "L6"); + const v1Text = lines("X", "L1", "L2", "L3", "L4", "BEFORE", "T", "AFTER", "L6"); store.record(PATH, v1Text); const { edits } = parsePatch("SWAP 5.=5:\n+MODEL"); @@ -214,8 +231,8 @@ describe("Recovery — colliding snapshot tags", () => { store.record(PATH, newer); // Live drifted away from both colliders, so recovery cannot shortcut - // via live==snapshot. The tag cannot name a unique base; it resolves - // to the most-recently recorded collider and 3-way merges from there. + // via live==snapshot. The tag cannot name a unique base; recovery uses + // the most-recently retained collider and maps its unchanged anchors. const currentText = `${newer}drifted trailer\n`; const recovered = new Recovery(store).tryRecover({ path: PATH, @@ -228,8 +245,7 @@ describe("Recovery — colliding snapshot tags", () => { }); it("still recovers when exactly one retained text carries the tag", () => { - // Same drift scenario with a single retained text for the tag: the - // plain 3-way merge path. + // Same drift scenario with a single retained text for the tag. const { older } = findCollidingTexts(); const store = new InMemorySnapshotStore(); const tag = store.record(PATH, older); From 485d207a7f31f321a2842caa8cddaf19dcb0fd94 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 17:33:15 +0200 Subject: [PATCH 07/28] fix(tui): prevented destructive screen replays during viewport resize - Refactor resize logic to keep forced renders on the viewport fast path mid-drag. - Eliminate redundant and visible full-transcript replays that previously occurred when forced renders interrupted a resize drag. - Ensure forced render intent is correctly folded into the final settle paint to maintain authoritative state. --- packages/tui/CHANGELOG.md | 4 ++ packages/tui/src/tui.ts | 25 +++++----- .../tui/test/resize-viewport-defer.test.ts | 48 +++++++++++++++++++ 3 files changed, 65 insertions(+), 12 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 3143afbfc..0007839a3 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed forced renders (tool finalization, `resetDisplay`, image reconciliation) landing during a resize drag preempting the alternate-screen viewport fast path: each one left the borrowed alt screen, erased native scrollback (ED3), and visibly replayed the whole transcript on the normal screen mid-drag — then the settle replayed it again. Forced intent now folds into the single authoritative settle paint. + ## [16.4.7] - 2026-07-12 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 885f07af2..56206cc61 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -2696,18 +2696,19 @@ export class TUI extends Container { // alternate screen to repaint the whole transcript on the normal // screen — then the next SIGWINCH re-enters the alt screen and paints // only the tail, so the block flashes in for one frame and vanishes. - // A forced render (tool finalization, reset, image reconciliation) must - // still preempt: it set #forceViewportRepaintOnNextRender via - // #prepareForcedRender and owns the next authoritative paint, so it falls - // through. A visible overlay composites over the transcript and needs the - // whole window, so it also falls through (overlay resizes are not on the - // drag-cost hot path). - if ( - this.#resizeViewportActive && - !this.#forceViewportRepaintOnNextRender && - this.#hasEverRendered && - this.#getTopmostVisibleOverlay() === undefined - ) { + // A FORCED render mid-drag (tool finalization, resetDisplay, image + // reconciliation) also stays on the fast path: preempting would leave + // the borrowed alternate screen and run the geometry-rebuild full paint + // on the normal screen — ED3 plus an O(history) replay that visibly + // scrolls the whole transcript through the viewport, once per forced + // render and once more at settle. The forced intent is not lost: the + // fast path consumes neither #forceViewportRepaintOnNextRender nor + // #clearScrollbackOnNextRender, and the settle's authoritative + // requestRender(true) honors both — same fold-into-the-settle contract + // as the multiplexer resize debounce. A visible overlay composites over + // the transcript and needs the whole window, so it falls through + // (overlay resizes are not on the drag-cost hot path). + if (this.#resizeViewportActive && this.#hasEverRendered && this.#getTopmostVisibleOverlay() === undefined) { this.#componentRenderTargets.clear(); this.#renderResizeViewport(width, height); return; diff --git a/packages/tui/test/resize-viewport-defer.test.ts b/packages/tui/test/resize-viewport-defer.test.ts index 622f8d212..3c9a1a5d4 100644 --- a/packages/tui/test/resize-viewport-defer.test.ts +++ b/packages/tui/test/resize-viewport-defer.test.ts @@ -341,6 +341,54 @@ describe("non-multiplexer resize viewport fast path", () => { }); }); + it("keeps a forced render mid-drag on the viewport fast path instead of a destructive normal-screen replay", async () => { + await withEnvPatch(NO_MULTIPLEXER_ENV, async () => { + const term = new VirtualTerminal(40, 10, 1000); + const { tui, scheduler } = makeTui(term); + try { + tui.start(); + await scheduler.flushImmediates(term); + + // One drag SIGWINCH enters the fast path and borrows the alt screen. + term.resize(60, 10); + await scheduler.flushImmediates(term); + expect(tui.resizeViewportActive).toBe(true); + + const baselineFull = tui.fullRedraws; + const baselinePaints = tui.resizeViewportPaints; + const writes = captureWrites(term); + + // A FORCED render lands mid-drag (tool finalization, resetDisplay, + // image reconciliation — routine during a live agent session). It must + // stay on the viewport fast path: preempting leaves the borrowed + // alternate screen and runs the geometry-rebuild full paint on the + // normal screen — ED3 plus an O(history) replay the user watches + // scroll through the viewport — and the settle then replays the whole + // transcript a second time. + tui.requestRender(true); + await scheduler.flushImmediates(term); + + // Still mid-drag, still on the alternate screen: a viewport-only + // paint, no full redraw, no scrollback erase, no alt-screen exit. + expect(tui.resizeViewportActive).toBe(true); + expect(tui.resizeViewportPaints).toBeGreaterThan(baselinePaints); + expect(tui.fullRedraws).toBe(baselineFull); + expect(writes.join("")).not.toContain(ALT_SCREEN_EXIT); + expect(eraseScrollbackCount(writes)).toBe(0); + + // The forced intent folds into the settle: exactly one authoritative + // full paint with exactly one ED3 once the drag goes quiet. + await scheduler.flushAll(term); + expect(tui.resizeViewportActive).toBe(false); + expect(tui.fullRedraws).toBe(baselineFull + 1); + expect(eraseScrollbackCount(writes)).toBe(1); + expect(visible(term).at(-1)).toBe("b14-y"); + } finally { + tui.stop(); + } + }); + }); + it("does not leave a pending settle paint after stop()", async () => { await withEnvPatch(NO_MULTIPLEXER_ENV, async () => { const term = new VirtualTerminal(40, 10, 1000); From 96cc9caa667c64bab19be881e8fce04c13833da5 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:27:24 +0200 Subject: [PATCH 08/28] feat(launch): enabled pty terminal rendering with standardized dimensions and serialization - Introduce `DAEMON_PTY_COLUMNS` and `DAEMON_PTY_ROWS` to `protocol.ts` for consistent PTY dimension management. - Implement `renderTerminalOutput` in `launch/terminal-output.ts` to utilize headless xterm for accurate terminal state replay. - Add `readTerminalRows` and `styleTerminalRow` in `tools/terminal-output.ts` to serialize terminal buffers while preserving safe ANSI styles. - Update `broker.ts` to use standardized PTY dimensions and ensure terminal output is properly sanitized and well-formed during stream processing. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/launch/broker.ts | 18 ++- packages/coding-agent/src/launch/protocol.ts | 4 + .../src/launch/terminal-output.ts | 46 ++++++ .../coding-agent/src/tools/terminal-output.ts | 143 ++++++++++++++++++ 5 files changed, 204 insertions(+), 8 deletions(-) create mode 100644 packages/coding-agent/src/launch/terminal-output.ts create mode 100644 packages/coding-agent/src/tools/terminal-output.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 96c51674b..8a90cde6c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -43,6 +43,7 @@ - Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately - Fixed the prewalk plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls - Fixed launch tool rendering stacking a stale pending header over a bare `✓ Launch` line and raw text: the tool now uses a merged registry renderer with one per-op status header (op, target, `state · pid · uptime` meta), stripped log cursor suffixes, capped collapsed log/list previews, and a launch tool glyph +- Fixed `launch logs` flattening PTY control sequences into repeated debugger frames: raw daemon output is now replayed through the shared xterm screen renderer used by Bash PTY mode, preserving cursor updates, colors, and text styles while model-facing log text remains sanitized. - Fixed confusing launch start/wait results when readiness timed out with the log pattern already matched (readiness needs log AND port): the result printed a contradictory `Ready: ` next to `Readiness timed out` without naming the failing condition. Daemon snapshots now carry the unmet conditions (`readyPending`), and start/wait results state exactly what never happened (e.g. `port 3100 on 127.0.0.1 never accepted connections`); the TUI shows a `waiting on port` badge on starting daemons - Fixed the in-process `stat` builtin mangling BSD-style invocations like `stat -f "%Sm %N" file` (macOS muscle memory): GNU `-f` means `--file-system`, so the format string was treated as a file operand — printing filesystem info for the real operands and erroring with `cannot read file system information for '%Sm %N'`. A `-f` whose format value contains `%` is now detected as BSD syntax and translated to the GNU equivalent (`%Sm`→`%y`, `%N`→`%n`, `%z`→`%s`, epoch/`S`-form times, owner/group/permission and `H`/`L` sub-field directives, `-L`/`-n`/`-q`/`-F` flag clusters, with `%n`/`%t` as literal newline/tab); directives with no GNU counterpart fail with a clear `unsupported BSD format directive` error - Fixed the remaining GNU-flavored shell builtins that broke under macOS/BSD muscle memory, using the same unambiguous-detection approach as the `stat` fix (only invocations that are invalid or nonsensical under GNU semantics are reinterpreted; unsupported BSD forms fail loudly instead of producing wrong output): `date -r ` formats the epoch when no such file exists (GNU `-r FILE` mtime preserved), signed `date -v±N` adjustments translate to `-d` relative dates and `-j` is accepted (`-j -f` strptime parse mode and field-set `-v` error clearly); `sed -i '' 's/…/…/' file` drops the BSD empty backup-suffix token instead of treating it as the script; `mktemp -t prefix` without X's creates `$TMPDIR/prefix.XXXXXXXXXX` (the GNU `too few X's` error path); `tail -r` reverses input by delegating to `tac` (with `-n`/`-c`/`-f` combinations erroring clearly); `find -E` maps to `-regextype posix-extended` ahead of the expression; `base64 -D` decodes as an alias of `-d`; and `ln -sfh` works via a `-h` alias of `--no-dereference` (clap's `-h` help short is dropped to match real GNU/BSD ln; `--help` unchanged) diff --git a/packages/coding-agent/src/launch/broker.ts b/packages/coding-agent/src/launch/broker.ts index c2bc6ef5c..9c8aa5db2 100644 --- a/packages/coding-agent/src/launch/broker.ts +++ b/packages/coding-agent/src/launch/broker.ts @@ -11,6 +11,8 @@ import { hasLiveDaemonProjectPresence } from "./presence"; import { DAEMON_IDLE_GRACE_ENV, DAEMON_PROJECT_DIR_ENV, + DAEMON_PTY_COLUMNS, + DAEMON_PTY_ROWS, DAEMON_RUNTIME_DIR_ENV, type DaemonOperation, type DaemonReadySpec, @@ -141,8 +143,7 @@ class DaemonLog { return new DaemonLog(logPath, previousPath, file, file.writer()); } - append(raw: string): string { - const text = sanitizeText(raw); + append(text: string): string { if (text.length === 0 || this.#closed) return text; const bytes = Buffer.byteLength(text, "utf8"); this.#queue = this.#queue.then(async () => { @@ -179,7 +180,7 @@ class DaemonLog { grep?: string, ): Promise { const [previous, current] = await Promise.all([fileTextSlice(previousPath, head), fileTextSlice(logPath, head)]); - let text = sanitizeText(`${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`); + let text = `${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`; if (grep) { let pattern: RegExp; try { @@ -189,7 +190,7 @@ class DaemonLog { } text = text .split("\n") - .filter(line => pattern.test(line)) + .filter(line => pattern.test(sanitizeText(line))) .join("\n"); } const options = { maxLines: lines, maxBytes: 256 * 1024 }; @@ -527,8 +528,8 @@ class DaemonBroker { command, cwd: record.spec.cwd, env: workerEnvFromParent({ TERM: "xterm-256color", ...record.spec.env }), - cols: 120, - rows: 40, + cols: DAEMON_PTY_COLUMNS, + rows: DAEMON_PTY_ROWS, shell, }, (error, chunk) => { @@ -626,9 +627,10 @@ class DaemonBroker { #onOutput(record: ManagedDaemon, generation: number, raw: string): void { if (generation !== record.generation) return; - const text = record.log?.append(raw) ?? sanitizeText(raw); + const output = raw.toWellFormed(); + const text = record.log?.append(output) ?? output; record.snapshot.outputBytes += Buffer.byteLength(text, "utf8"); - this.#trackOutput(record, generation, text); + this.#trackOutput(record, generation, sanitizeText(text)); } async #readDetachedOutput(record: ManagedDaemon, generation: number): Promise { diff --git a/packages/coding-agent/src/launch/protocol.ts b/packages/coding-agent/src/launch/protocol.ts index 97bf0631c..0ad3e31cf 100644 --- a/packages/coding-agent/src/launch/protocol.ts +++ b/packages/coding-agent/src/launch/protocol.ts @@ -4,6 +4,10 @@ /** Hidden CLI selector used to re-enter the daemon broker worker. */ export const DAEMON_BROKER_WORKER_ARG = "__omp_worker_daemon_broker"; +/** Fixed dimensions negotiated with every supervised PTY. */ +export const DAEMON_PTY_COLUMNS = 120; +export const DAEMON_PTY_ROWS = 40; + /** Environment key carrying the broker's canonical project directory. */ export const DAEMON_PROJECT_DIR_ENV = "OMP_DAEMON_PROJECT_DIR"; diff --git a/packages/coding-agent/src/launch/terminal-output.ts b/packages/coding-agent/src/launch/terminal-output.ts new file mode 100644 index 000000000..da75c5c6c --- /dev/null +++ b/packages/coding-agent/src/launch/terminal-output.ts @@ -0,0 +1,46 @@ +import { logger } from "@oh-my-pi/pi-utils"; +import xterm, { type Terminal as XtermTerminal } from "@xterm/headless"; +import { readTerminalRows } from "../tools/terminal-output"; +import { DAEMON_PTY_COLUMNS, DAEMON_PTY_ROWS } from "./protocol"; + +const VIRTUAL_SCROLLBACK_ROWS = 4_096; + +/** Controls which virtual terminal rows a launch log exposes. */ +export interface TerminalOutputOptions { + head: boolean; + maxRows: number; +} + +function writeTerminal(terminal: XtermTerminal, output: string): Promise { + const { promise, resolve } = Promise.withResolvers(); + terminal.write(output, resolve); + return promise; +} + +/** Replays daemon bytes with the same xterm screen renderer used by PTY mode. */ +export async function renderTerminalOutput( + output: string, + options: TerminalOutputOptions, +): Promise { + if (!output) return []; + const maxRows = Math.max(1, Math.floor(options.maxRows)); + const terminal = new xterm.Terminal({ + cols: DAEMON_PTY_COLUMNS, + rows: DAEMON_PTY_ROWS, + scrollback: Math.max(VIRTUAL_SCROLLBACK_ROWS, maxRows), + allowProposedApi: true, + }); + try { + await writeTerminal(terminal, output); + const rows = readTerminalRows(terminal, 0, terminal.buffer.active.length); + while (rows.at(-1) === "") rows.pop(); + return options.head ? rows.slice(0, maxRows) : rows.slice(-maxRows); + } catch (error) { + logger.debug("Failed to render launch terminal output", { + error: error instanceof Error ? error.message : String(error), + }); + return undefined; + } finally { + terminal.dispose(); + } +} diff --git a/packages/coding-agent/src/tools/terminal-output.ts b/packages/coding-agent/src/tools/terminal-output.ts new file mode 100644 index 000000000..e77dc1f4e --- /dev/null +++ b/packages/coding-agent/src/tools/terminal-output.ts @@ -0,0 +1,143 @@ +import { sanitizeText } from "@oh-my-pi/pi-utils"; +import type { Terminal as XtermTerminal } from "@xterm/headless"; + +const RESET = "\x1b[0m"; +const SGR = /\x1b\[([0-9;]*)m/g; + +interface TerminalCell { + getChars(): string; + getWidth(): number; + getFgColor(): number; + getBgColor(): number; + isBold(): number; + isDim(): number; + isItalic(): number; + isUnderline(): number; + isInverse(): number; + isStrikethrough(): number; + isOverline(): number; + isFgRGB(): boolean; + isBgRGB(): boolean; + isFgPalette(): boolean; + isBgPalette(): boolean; +} + +function addColor(codes: number[], cell: TerminalCell, foreground: boolean): void { + const rgb = foreground ? cell.isFgRGB() : cell.isBgRGB(); + const palette = foreground ? cell.isFgPalette() : cell.isBgPalette(); + if (!rgb && !palette) return; + + const color = foreground ? cell.getFgColor() : cell.getBgColor(); + codes.push(foreground ? 38 : 48); + if (rgb) { + codes.push(2, (color >> 16) & 0xff, (color >> 8) & 0xff, color & 0xff); + } else { + codes.push(5, color); + } +} + +function cellStyle(cell: TerminalCell): string { + const codes: number[] = []; + if (cell.isBold() !== 0) codes.push(1); + if (cell.isDim() !== 0) codes.push(2); + if (cell.isItalic() !== 0) codes.push(3); + if (cell.isUnderline() !== 0) codes.push(4); + if (cell.isInverse() !== 0) codes.push(7); + if (cell.isStrikethrough() !== 0) codes.push(9); + if (cell.isOverline() !== 0) codes.push(53); + addColor(codes, cell, true); + addColor(codes, cell, false); + return codes.length > 0 ? `\x1b[${codes.join(";")}m` : ""; +} + +function isSafeStyle(codes: readonly number[]): boolean { + let index = 0; + while (index < codes.length) { + const code = codes[index++]; + if (code === 1 || code === 2 || code === 3 || code === 4 || code === 7 || code === 9 || code === 53) continue; + if (code !== 38 && code !== 48) return false; + const mode = codes[index++]; + if (mode === 5) { + const color = codes[index++]; + if (color === undefined || color < 0 || color > 255) return false; + continue; + } + if (mode !== 2) return false; + for (let channel = 0; channel < 3; channel++) { + const color = codes[index++]; + if (color === undefined || color < 0 || color > 255) return false; + } + } + return true; +} + +/** Applies the active tool-output color while preserving safe styles from a virtual terminal row. */ +export function styleTerminalRow(row: string, baseForeground: string): string { + let output = baseForeground; + let offset = 0; + let hasText = false; + for (const match of row.matchAll(SGR)) { + const index = match.index ?? 0; + const text = sanitizeText(row.slice(offset, index)); + output += text; + hasText ||= text.length > 0; + + const codes = match[1].split(";").map(Number); + if (match[1] === "0") output += `${RESET}${baseForeground}`; + else if (codes.length > 0 && codes.every(Number.isInteger) && isSafeStyle(codes)) output += match[0]; + offset = index + match[0].length; + } + const text = sanitizeText(row.slice(offset)); + output += text; + hasText ||= text.length > 0; + return hasText ? `${output}${RESET}` : ""; +} + +/** Reads terminal screen rows as sanitized text plus only the styles the TUI may replay. */ +export function readTerminalRows(terminal: XtermTerminal, startRow: number, rowCount: number): string[] { + const buffer = terminal.buffer.active; + const reusableCell = buffer.getNullCell(); + const rows: string[] = []; + const endRow = Math.min(buffer.length, Math.max(0, startRow) + Math.max(0, rowCount)); + + for (let row = Math.max(0, startRow); row < endRow; row++) { + const line = buffer.getLine(row); + if (!line) { + rows.push(""); + continue; + } + + const cells: Array<{ chars: string; style: string }> = []; + let lastContent = -1; + for (let column = 0; column < line.length; ) { + const cell = line.getCell(column, reusableCell); + if (!cell) break; + const chars = cell.getChars(); + const width = Math.max(1, cell.getWidth()); + if (chars) { + cells.push({ chars, style: cellStyle(cell) }); + if (chars !== " ") lastContent = cells.length - 1; + } + column += width; + } + + if (lastContent < 0) { + rows.push(""); + continue; + } + + let rendered = ""; + let previousStyle: string | undefined; + for (let index = 0; index <= lastContent; index++) { + const cell = cells[index]!; + if (cell.style !== previousStyle) { + rendered += `${RESET}${cell.style}`; + previousStyle = cell.style; + } + rendered += cell.chars; + } + rows.push(rendered); + } + + return rows; +} From 0f9a3015367ae31c1b79fc264194f1a2975860bb Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:27:24 +0200 Subject: [PATCH 09/28] fix(harbor): fixed Disable dev console mirroring in bun server to prevent AbortError crashes during hot reloads - Disable dev console mirroring in bun server to prevent AbortError crashes during hot reloads. - Verify runner process liveness with process.kill(pid, 0) instead of relying solely on recorded job status. - Force exit marker for jobs marked as running that no longer have an active system process. --- packages/harbor-manager/src/server.ts | 21 +++++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index 755e052b7..e9af7650b 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -182,7 +182,11 @@ export class ManagerServer { // Bun bundles the dashboard (React + TSX) from the HTML import and // serves it on the same port as the API — one process, no Vite. routes: { "/": indexHtml }, - development: process.env.NODE_ENV !== "production" && { hmr: true, console: true }, + // Only `hmr`: Bun's `console: true` mirror opens a server-read stream + // over the dev client that a `--hot` reload's force-close tears down + // mid-read, surfacing an unhandled `AbortError: ERR_STREAM_RELEASE_LOCK` + // that crashes the process. + development: process.env.NODE_ENV !== "production" && { hmr: true }, fetch: request => this.#route(request), }); return this.#server; @@ -380,9 +384,22 @@ export class ManagerServer { if (!run) throw new Error(`run ${jobName} not found`); if (run.benchmark !== "harbor") throw new Error(`resume supports only harbor runs (${jobName} is ${run.benchmark})`); - if (this.#children.has(jobName) || run.status === "running") { + // Trust liveness, not the recorded status: a runner killed while a + // previous server instance owned it leaves a stale `running` row with a + // dead (or null) pid and nobody to fire markExit. + const pidAlive = (pid: number | null): boolean => { + if (pid == null) return false; + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } + }; + if (this.#children.has(jobName) || (run.status === "running" && pidAlive(run.pid))) { throw new Error(`run ${jobName} is already running`); } + if (run.status === "running") this.#store.markExit(jobName, null, true); const jobDir = path.join(this.jobsDir, jobName); if (!fs.existsSync(path.join(jobDir, "config.json"))) { throw new Error(`${jobName} has no harbor config.json to resume from`); From 6fcb1b3008a45bfc96f5543c8c435f2b307ac801 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:27:25 +0200 Subject: [PATCH 10/28] chore(test): synchronized test expectations with output and cleanup formatting - Updated `hashline` test expectations to match new recovery message strings. - Initialized settings in `interactive-mode-status` tests to prevent state leakage. - Added ANSI color code verification to `launch` tool tests to align with updated output. - Cleaned up whitespace in `tan-context-switch.md` system prompt. --- .../coding-agent/src/prompts/system/tan-context-switch.md | 2 +- packages/coding-agent/test/core/hashline.test.ts | 4 ++-- packages/coding-agent/test/interactive-mode-status.test.ts | 6 +++++- packages/coding-agent/test/tools/launch.test.ts | 3 ++- 4 files changed, 10 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 3e41f52a7..8265640bf 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -369,7 +369,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(finalLines).toContain("L8"); const text = result.content[0]?.type === "text" ? result.content[0].text : ""; - expect(text).toMatch(/Recovered from a stale file hash using a previous read snapshot/); + expect(text).toMatch(/Recovered by remapping stale line anchors to unchanged current lines/); }); }); @@ -437,7 +437,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(finalLines).toContain("GAMMA"); expect(finalLines).not.toContain("gamma"); const text = result.content[0]?.type === "text" ? result.content[0].text : ""; - expect(text).toMatch(/Recovered from a stale file hash using a previous read snapshot/); + expect(text).toMatch(/Recovered by remapping stale line anchors to unchanged current lines/); }); }); diff --git a/packages/coding-agent/test/interactive-mode-status.test.ts b/packages/coding-agent/test/interactive-mode-status.test.ts index 2a2bff09c..542564c52 100644 --- a/packages/coding-agent/test/interactive-mode-status.test.ts +++ b/packages/coding-agent/test/interactive-mode-status.test.ts @@ -1,5 +1,6 @@ import { beforeAll, describe, expect, test, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; import { UiHelpers } from "@oh-my-pi/pi-coding-agent/modes/utils/ui-helpers"; @@ -64,7 +65,10 @@ function createInitialRenderHarness(): { ctx: InteractiveModeContext; helpers: U describe("InteractiveMode.showStatus", () => { beforeAll(async () => { - // showStatus uses the global theme instance + // showStatus uses the global theme instance; renderInitialMessages reads + // the global Settings (display.collapseCompacted). + resetSettingsForTest(); + await Settings.init({ inMemory: true }); await initTheme(); }); diff --git a/packages/coding-agent/test/tools/launch.test.ts b/packages/coding-agent/test/tools/launch.test.ts index eb5c95f3e..84b6bd48c 100644 --- a/packages/coding-agent/test/tools/launch.test.ts +++ b/packages/coding-agent/test/tools/launch.test.ts @@ -59,7 +59,7 @@ describe("daemon broker", () => { `process.stdin.setRawMode?.(true); process.stdin.setEncoding("utf8"); process.stdin.resume(); -process.stdout.write("READY\\n"); +process.stdout.write("\\x1b[1;32mREADY\\x1b[0m\\n"); process.stdin.on("data", chunk => process.stdout.write("INPUT:" + JSON.stringify(chunk) + "\\n")); setInterval(() => {}, 1000); `, @@ -114,6 +114,7 @@ setInterval(() => {}, 1000); expect(logs.op).toBe("logs"); if (logs.op !== "logs") throw new Error("unexpected logs result"); expect(logs.text).toContain("READY"); + expect(logs.text).toContain("\x1b[1;32mREADY\x1b[0m"); expect(logs.text).toContain('INPUT:"run\\r"'); const stopped = await first.request({ op: "stop", name: "debugger", timeoutMs: 2_000 }); From acc0211cfdcd60346299596103387bce5e189d5d Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:27:26 +0200 Subject: [PATCH 11/28] feat(tools): integrated terminal rendering into launch tool output - Updated `launch` tool to utilize `renderTerminalOutput` for improved log formatting. - Added `terminalRows` to `LaunchToolDetails` to support structured virtual terminal rendering. - Refactored `BashInteractiveOverlayComponent` to consolidate terminal line reading and styling logic. - Applied sanitization to log text in `toolContent` to prevent rendering issues. - Integrated `styleTerminalRow` in the `launch` tool renderer to maintain consistent UI theme application. --- .../src/tools/bash-interactive.ts | 13 +++---- packages/coding-agent/src/tools/launch.ts | 36 ++++++++++++++----- 2 files changed, 33 insertions(+), 16 deletions(-) diff --git a/packages/coding-agent/src/tools/bash-interactive.ts b/packages/coding-agent/src/tools/bash-interactive.ts index fc39cb0c8..8618146fe 100644 --- a/packages/coding-agent/src/tools/bash-interactive.ts +++ b/packages/coding-agent/src/tools/bash-interactive.ts @@ -19,6 +19,7 @@ import { OutputSink, type OutputSummary } from "../session/streaming-output"; import { sanitizeWithOptionalSixelPassthrough } from "../utils/sixel"; import { resolveOutputMaxColumns, resolveOutputSinkHeadBytes } from "./output-meta"; import { formatStatusIcon, replaceTabs } from "./render-utils"; +import { readTerminalRows, styleTerminalRow } from "./terminal-output"; export interface BashInteractiveResult extends OutputSummary { exitCode: number | undefined; @@ -225,14 +226,10 @@ class BashInteractiveOverlayComponent implements Component { #readViewport(innerWidth: number, maxContentRows: number): string[] { this.#terminal.resize(innerWidth, maxContentRows); - const buffer = this.#terminal.buffer.active; - const viewportY = buffer.viewportY; - const visibleLines: string[] = []; - for (let i = 0; i < maxContentRows; i++) { - const line = buffer.getLine(viewportY + i)?.translateToString(true) ?? ""; - visibleLines.push(truncateToWidth(replaceTabs(sanitizeText(line)), innerWidth)); - } - return visibleLines; + const viewportY = this.#terminal.buffer.active.viewportY; + return readTerminalRows(this.#terminal, viewportY, maxContentRows).map(line => + truncateToWidth(styleTerminalRow(line, this.uiTheme.getFgAnsi("toolOutput")), innerWidth), + ); } render(width: number): readonly string[] { const safeWidth = Math.max(20, width); diff --git a/packages/coding-agent/src/tools/launch.ts b/packages/coding-agent/src/tools/launch.ts index 879c1555e..31c628bbf 100644 --- a/packages/coding-agent/src/tools/launch.ts +++ b/packages/coding-agent/src/tools/launch.ts @@ -8,11 +8,12 @@ import type { import type { ToolExample } from "@oh-my-pi/pi-ai"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; -import { prompt } from "@oh-my-pi/pi-utils"; +import { prompt, sanitizeText } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { daemonClientForProject } from "../launch/client"; import type { DaemonOperation, DaemonRpcResult, DaemonSnapshot, DaemonSpec, DaemonState } from "../launch/protocol"; +import { renderTerminalOutput } from "../launch/terminal-output"; import type { Theme, ThemeColor } from "../modes/theme/theme"; import launchDescription from "../prompts/tools/launch.md" with { type: "text" }; import { renderStatusLine } from "../tui"; @@ -32,6 +33,7 @@ import { TRUNCATE_LENGTHS, truncateToWidth, } from "./render-utils"; +import { styleTerminalRow } from "./terminal-output"; import { ToolError } from "./tool-errors"; const launchSchema = type({ @@ -92,6 +94,8 @@ export interface LaunchToolDetails { timedOut?: boolean; /** logs: daemon lifecycle state at read time. */ state?: DaemonState; + /** logs: virtual terminal rows for display; model-facing content remains sanitized text. */ + terminalRows?: string[]; /** wait: output line that satisfied the pattern. */ matched?: string; /** describe: immutable launch spec backing the command/cwd detail lines. */ @@ -243,8 +247,10 @@ function toolContent(result: DaemonRpcResult, params: LaunchParams): string { return result.daemons.length ? result.daemons.map(daemon => `- ${daemonLabel(daemon)}`).join("\n") : "No daemons."; - case "logs": - return `${result.text}${result.text && !result.text.endsWith("\n") ? "\n" : ""}[${result.name}: ${result.state}; cursor=${result.cursor}${result.timedOut ? "; follow timed out" : ""}]`; + case "logs": { + const text = sanitizeText(result.text); + return `${text}${text && !text.endsWith("\n") ? "\n" : ""}[${result.name}: ${result.state}; cursor=${result.cursor}${result.timedOut ? "; follow timed out" : ""}]`; + } case "wait": { const lines = [daemonLabel(result.daemon)]; if (result.matched) lines.push(`Matched: ${result.matched}`); @@ -270,14 +276,25 @@ function toolContent(result: DaemonRpcResult, params: LaunchParams): string { } } -function toolDetails(result: DaemonRpcResult): LaunchToolDetails { +async function toolDetails(result: DaemonRpcResult, params: LaunchParams): Promise { switch (result.op) { case "start": return { op: "start", daemon: result.daemon, timedOut: result.readyTimedOut }; case "list": return { op: "list", daemons: result.daemons }; - case "logs": - return { op: "logs", cursor: result.cursor, timedOut: result.timedOut, state: result.state }; + case "logs": { + const terminalRows = await renderTerminalOutput(result.text, { + head: params.head ?? false, + maxRows: Math.min(1_000, Math.floor(params.lines ?? 100)), + }); + return { + op: "logs", + cursor: result.cursor, + timedOut: result.timedOut, + state: result.state, + terminalRows, + }; + } case "wait": return { op: "wait", daemon: result.daemon, timedOut: result.timedOut, matched: result.matched }; case "send": @@ -372,7 +389,7 @@ export class LaunchTool implements AgentTool Date: Mon, 13 Jul 2026 18:27:26 +0200 Subject: [PATCH 12/28] test(launch): validated terminal screen replay rendering and state - Verify terminal screen row replay correctly handles cursor rewrites. - Ensure final text color and weight are preserved during terminal output rendering. - Assert that superseded text is excluded from the rendered output. --- .../test/tools/launch-renderer.test.ts | 31 +++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/packages/coding-agent/test/tools/launch-renderer.test.ts b/packages/coding-agent/test/tools/launch-renderer.test.ts index e9c2723b3..9dfb27f47 100644 --- a/packages/coding-agent/test/tools/launch-renderer.test.ts +++ b/packages/coding-agent/test/tools/launch-renderer.test.ts @@ -6,6 +6,7 @@ */ import { describe, expect, it } from "bun:test"; import type { DaemonSnapshot } from "@oh-my-pi/pi-coding-agent/launch/protocol"; +import { renderTerminalOutput } from "@oh-my-pi/pi-coding-agent/launch/terminal-output"; import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { type LaunchToolDetails, launchToolRenderer } from "@oh-my-pi/pi-coding-agent/tools/launch"; import { toolRenderers } from "@oh-my-pi/pi-coding-agent/tools/renderers"; @@ -83,6 +84,36 @@ describe("launchToolRenderer", () => { expect(rendered.some(line => line.includes("[web: running"))).toBe(false); }); + it("replays terminal screen rows so cursor rewrites retain their final color and weight", async () => { + const terminalRows = await renderTerminalOutput("\x1b[1;31mold\x1b[0m\r\x1b[1;32mready\x1b[0m\x1b[K", { + head: false, + maxRows: 10, + }); + if (terminalRows === undefined) throw new Error("terminal replay failed"); + + const uiTheme = await theme(); + const component = launchToolRenderer.renderResult( + { + content: [{ type: "text", text: "ready\n[web: running; cursor=2210]" }], + details: { + op: "logs", + cursor: 2210, + timedOut: false, + state: "running", + terminalRows, + } satisfies LaunchToolDetails, + }, + { expanded: false, isPartial: false }, + uiTheme, + { op: "logs", name: "web" }, + ); + const raw = component.render(200).join("\n"); + const plain = Bun.stripANSI(raw); + expect(plain).toContain("ready"); + expect(plain).not.toContain("old"); + expect(raw).toContain("\x1b[1;38;5;2mready"); + }); + it("caps a collapsed list to the preview item limit with a more-items row", async () => { const uiTheme = await theme(); const daemons = Array.from({ length: 11 }, (_, i) => daemon({ name: `svc-${i}`, id: `d-${i}` })); From a5673c90f823915a98a5e8323b806c2ab778b0e9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:43:52 +0200 Subject: [PATCH 13/28] feat(ai): removed legacy Google interactions routing from AI providers - Removed the Google Interactions transport and deleted interaction-specific request options from the shared AI stream typing/API surface. - Simplified Google provider routing to eliminate interactions auto-selection logic and keep `streamGoogle` on the `:streamGenerateContent` path. - Updated Vertex request handling to use resolved stream hosts without `/interactions`/`Api-Revision` and removed related interaction constants. - Deleted obsolete Interactions tests and updated remaining Google stream tests to no longer reference `useInteractionsApi`/`storeInteraction`/`previousInteractionId`. --- bun.lock | 16 +- package.json | 2 +- packages/ai/CHANGELOG.md | 9 + packages/ai/src/providers/google-auth.ts | 20 - .../ai/src/providers/google-interactions.ts | 753 ------------------ packages/ai/src/providers/google-shared.ts | 13 - packages/ai/src/providers/google-vertex.ts | 178 ++--- packages/ai/src/providers/google.ts | 62 +- packages/ai/src/stream.ts | 3 - packages/ai/src/types.ts | 16 - .../test/google-empty-response-retry.test.ts | 7 +- packages/ai/test/google-interactions.test.ts | 511 ------------ packages/ai/test/google-service-tier.test.ts | 7 +- packages/ai/test/google-system-prompt.test.ts | 2 - packages/ai/test/issue-1270-repro.test.ts | 2 - packages/coding-agent/src/sdk.ts | 2 +- .../coding-agent/src/session/agent-session.ts | 6 +- .../agent-session-prune-persistence.test.ts | 4 +- 18 files changed, 99 insertions(+), 1514 deletions(-) delete mode 100644 packages/ai/src/providers/google-interactions.ts delete mode 100644 packages/ai/test/google-interactions.test.ts diff --git a/bun.lock b/bun.lock index ed47a4f51..e57ea34e3 100644 --- a/bun.lock +++ b/bun.lock @@ -164,8 +164,6 @@ "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", "@types/react-dom": "^19.1.0", - "@vitejs/plugin-react": "^5.0.4", - "vite": "catalog:", }, }, "packages/hashline": { @@ -355,7 +353,7 @@ "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.4.16", + "@biomejs/biome": "^2.5", "@bufbuild/protobuf": "^2.12.0", "@bufbuild/protoc-gen-es": "^2.12.0", "@huggingface/transformers": "^4.2.0", @@ -473,10 +471,6 @@ "@babel/plugin-syntax-jsx": ["@babel/plugin-syntax-jsx@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-TSu8+mHCoEaaCDEZ0I3+6mvTBYR4PCxQwf2z9/r5Tbztv6NaLR3B9thGTTxX2WGuGHJqRiAbKPeGTJ5XWXVg6A=="], - "@babel/plugin-transform-react-jsx-self": ["@babel/plugin-transform-react-jsx-self@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-TL0hMc9xzy86VD31nUiwzd5otRAcyEPcsegCxolO0PvcXuH1v0kECe/UIznYFihpkvU5wg/jk4v0TTEFfm53fw=="], - - "@babel/plugin-transform-react-jsx-source": ["@babel/plugin-transform-react-jsx-source@7.29.7", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.29.7" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-06IyK09H3wi4cGbhDBwp5gUGo0IKtnYa8tyTiephirPCK6fbobVGiXMMI5zLQ4aKEYP3wZ3ArU44o+8KMrSG/Q=="], - "@babel/template": ["@babel/template@7.29.7", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/types": "^7.29.7" } }, "sha512-puq+Gf35oI24FeN11LkoUQFqv9uwNeWpxXZi/Ji3rRIoKAzKnxRaZ+Gkj0vKS9ZCiTESfng1N9LyOyXvo+m+Gg=="], "@babel/traverse": ["@babel/traverse@7.29.7", "", { "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/generator": "^7.29.7", "@babel/helper-globals": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/template": "^7.29.7", "@babel/types": "^7.29.7", "debug": "^4.3.1" } }, "sha512-EhlfNQtZ+NK22w5BM61ciuiq1m58ed33Wr1Xan//ZRTy6hgjnwyCffRYwzsGXdASJSUJ1guZILsErh1eQcl+zw=="], @@ -873,7 +867,7 @@ "@rolldown/binding-win32-x64-msvc": ["@rolldown/binding-win32-x64-msvc@1.1.5", "", { "os": "win32", "cpu": "x64" }, "sha512-tTZuDBPw85tEN5PQi1pnEBzDy0Z49HtScLAbD5t6hyeU92A95pRWaSMw1GZZi/RwgSgUIl0xrSlXIT/9QzvYSA=="], - "@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.0-rc.3", "", {}, "sha512-eybk3TjzzzV97Dlj5c+XrBFW57eTNhzod66y9HrBlzJ6NsCrWCp/2kaPS3K9wJmurBC0Tdw4yPjXKZqlznim3Q=="], + "@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="], "@sindresorhus/is": ["@sindresorhus/is@4.6.0", "", {}, "sha512-t09vSN3MdfsyCHoFcTRCH/iUtG7OJ0CsjzB8cjAmKc/va/kIgeDI/TxsigdncE/4be734m0cvIYwNaV4i2XqAw=="], @@ -959,8 +953,6 @@ "@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, "sha512-PJFXFS4ZJKiJ9Qiuix6Dz/OwEIqHD7Dme1UwZhTK11vR+5dqW2ACbdndWQexBzCx+CPuMe5WBYQWCsFyGlQLlQ=="], - "@vitejs/plugin-react": ["@vitejs/plugin-react@5.2.0", "", { "dependencies": { "@babel/core": "^7.29.0", "@babel/plugin-transform-react-jsx-self": "^7.27.1", "@babel/plugin-transform-react-jsx-source": "^7.27.1", "@rolldown/pluginutils": "1.0.0-rc.3", "@types/babel__core": "^7.20.5", "react-refresh": "^0.18.0" }, "peerDependencies": { "vite": "^4.2.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0" } }, "sha512-YmKkfhOAi3wsB1PhJq5Scj3GXMn3WvtQ/JC0xoopuHoXSdmtdStOpFrYaT1kie2YgFBcIe64ROzMYRjCrYOdYw=="], - "@xmldom/xmldom": ["@xmldom/xmldom@0.8.13", "", {}, "sha512-KRYzxepc14G/CEpEGc3Yn+JKaAeT63smlDr+vjB8jRfgTBBI9wRj/nkQEO+ucV8p8I9bfKLWp37uHgFrbntPvw=="], "@xterm/headless": ["@xterm/headless@6.0.0", "", {}, "sha512-5Yj1QINYCyzrZtf8OFIHi47iQtI+0qYFPHmouEfG8dHNxbZ9Tb9YGSuLcsEwj9Z+OL75GJqPyJbyoFer80a2Hw=="], @@ -1385,8 +1377,6 @@ "react-dom": ["react-dom@19.2.7", "", { "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { "react": "^19.2.7" } }, "sha512-t0BRVXvbiE/o20Hfw669rLbMCDWtYZLvmJigy2f0MxsXF+71pxhR3xOkspmsO8h3ZlNzyibAmtCa3l4lYKk6gQ=="], - "react-refresh": ["react-refresh@0.18.0", "", {}, "sha512-QgT5//D3jfjJb6Gsjxv0Slpj23ip+HtOpnNgnb2S5zU3CB26G/IDPGoy4RJB42wzFE46DRsstbW6tKHoKbhAxw=="], - "readable-stream": ["readable-stream@3.6.2", "", { "dependencies": { "inherits": "^2.0.3", "string_decoder": "^1.1.1", "util-deprecate": "^1.0.1" } }, "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA=="], "regexp-tree": ["regexp-tree@0.1.27", "", { "bin": { "regexp-tree": "bin/regexp-tree" } }, "sha512-iETxpjK6YoRWJG5o6hXLwvjYAoW+FEZn9os0PD/b6AP6xQwsa/Y7lCVgIixBbUPMfhu+i2LtdeAqVTgGlQarfA=="], @@ -1639,8 +1629,6 @@ "roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="], - "rolldown/@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="], - "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], "wrap-ansi/string-width": ["string-width@8.2.2", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-GaPUh5gfdrYzqeVNZvUfT23vYYxXzKYidUcnMtJg/3rxRV63EFZy3k6xfKlmfeJD0176lnUV/Usr3XcwSvFzpg=="], diff --git a/package.json b/package.json index 7e88c3232..31b1e8499 100644 --- a/package.json +++ b/package.json @@ -19,7 +19,7 @@ "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.4.16", + "@biomejs/biome": "^2.5", "@bufbuild/protobuf": "^2.12.0", "@bufbuild/protoc-gen-es": "^2.12.0", "@huggingface/transformers": "^4.2.0", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 927ed24cb..03fa9035e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,15 @@ ## [Unreleased] +### Changed + +- Switched Google and Google Vertex providers to always use `streamGenerateContent` requests + +### Removed + +- Removed automatic `/interactions` chaining for follow-up turns in Google provider calls +- Removed `useInteractionsApi`, `storeInteraction`, and `previousInteractionId` from stream options + ### Fixed - Fixed empty provider responses (e.g. "Cloud Code Assist API returned an empty response") being classified as non-retryable: `ProviderResponseError` with kind `empty-body` now carries the transient flag, so session retry and configured model-fallback chains engage instead of hard-failing the turn diff --git a/packages/ai/src/providers/google-auth.ts b/packages/ai/src/providers/google-auth.ts index 7aabaf5cc..740a253cf 100644 --- a/packages/ai/src/providers/google-auth.ts +++ b/packages/ai/src/providers/google-auth.ts @@ -13,7 +13,6 @@ */ import { Buffer } from "node:buffer"; -import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils"; @@ -329,22 +328,3 @@ export function __resetVertexTokenCache(): void { tokenCache.clear(); inflight.clear(); } - -/** - * Sync best-effort probe for a usable Vertex bearer credential source — an explicit access-token - * env var, `GOOGLE_APPLICATION_CREDENTIALS`, a user ADC file, or a GCP runtime whose metadata - * server can mint ADC (GCE/Cloud Run/App Engine/Functions). Lets callers prefer the bearer - * Interactions transport only when ADC is actually reachable, without paying the async - * metadata-probe cost for API-key-only setups. - */ -export function hasVertexBearerCredentialsHint(): boolean { - if (Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN) return true; - if (Bun.env.GOOGLE_APPLICATION_CREDENTIALS) return true; - // GCP-hosted runtimes expose ADC via the metadata server; these env vars mark those runtimes. - if (Bun.env.K_SERVICE || Bun.env.FUNCTION_TARGET || Bun.env.GAE_ENV || Bun.env.GCE_METADATA_HOST) return true; - try { - return fs.existsSync(userAdcPath()); - } catch { - return false; - } -} diff --git a/packages/ai/src/providers/google-interactions.ts b/packages/ai/src/providers/google-interactions.ts deleted file mode 100644 index 3f5612809..000000000 --- a/packages/ai/src/providers/google-interactions.ts +++ /dev/null @@ -1,753 +0,0 @@ -import { parseGeminiModel } from "@oh-my-pi/pi-catalog/identity"; -import { calculateCost } from "@oh-my-pi/pi-catalog/models"; -import { fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils"; -import * as AIError from "../error"; -import type { - AssistantMessage, - Context, - FetchImpl, - ImageContent, - Message, - Model, - ProviderSessionState, - TextContent, - ToolCall, - Usage, -} from "../types"; -import { shouldSendServiceTier } from "../types"; -import { normalizeSystemPrompts } from "../utils"; -import { AssistantMessageEventStream } from "../utils/event-stream"; -import { convertTools, type GoogleSharedStreamOptions, type GoogleThinkingLevel } from "./google-shared"; - -type GoogleInteractionsApi = "google-generative-ai" | "google-vertex"; -type GoogleInteractionsModel = Model; -type GoogleOptions = GoogleSharedStreamOptions; - -const GOOGLE_INTERACTIONS_STATE_KEY = "google-interactions-state"; - -/** Provider session state storing the last Gemini Interactions response id. */ -export interface GoogleInteractionsProviderSessionState extends ProviderSessionState { - lastInteractionId?: string; -} - -/** Conversation anchor for continuing an Interactions turn from a prior assistant response. */ -export interface InteractionAnchor { - id?: string; - messageIndex?: number; -} - -type InteractionContent = { type: "text"; text: string } | { type: "image"; data: string; mime_type: string }; - -interface InteractionUserInputStep { - type: "user_input"; - content: InteractionContent[]; -} - -interface InteractionModelOutputStep { - type: "model_output"; - content: InteractionContent[]; -} - -interface InteractionFunctionCallStep { - type: "function_call"; - id: string; - name: string; - arguments: Record; -} - -interface InteractionFunctionResultStep { - type: "function_result"; - name: string; - call_id: string; - result: InteractionContent[]; - is_error?: boolean; -} - -interface InteractionThoughtStep { - type: "thought"; - summary?: InteractionContent[]; - signature?: string; -} - -interface InteractionThoughtSummaryDelta { - type: "thought_summary"; - content?: InteractionContent; -} - -interface InteractionThoughtSignatureDelta { - type: "thought_signature"; - signature?: string; -} - -interface InteractionArgumentsDelta { - type: "arguments_delta"; - arguments?: string; -} - -interface PendingInteractionToolCall { - id: string; - name: string; - argumentsText: string; - argumentsObject: Record; -} - -type InteractionInputStep = - | InteractionUserInputStep - | InteractionModelOutputStep - | InteractionFunctionCallStep - | InteractionFunctionResultStep; - -type InteractionStep = - | InteractionModelOutputStep - | InteractionFunctionCallStep - | InteractionFunctionResultStep - | InteractionThoughtStep - | InteractionUserInputStep; - -type InteractionThinkingLevel = "minimal" | "low" | "medium" | "high"; - -interface InteractionGenerationConfig { - temperature?: number; - top_p?: number; - top_k?: number; - min_p?: number; - presence_penalty?: number; - frequency_penalty?: number; - repetition_penalty?: number; - max_output_tokens?: number; - thinking_level?: InteractionThinkingLevel; - thinking_budget?: number; -} - -interface GoogleInteractionRequest { - model: string; - input: InteractionInputStep[]; - stream: true; - previous_interaction_id?: string; - system_instruction?: string; - tools?: { functionDeclarations: Record[] }[]; - store?: boolean; - generation_config?: InteractionGenerationConfig; - service_tier?: string; -} - -interface InteractionUsage { - total_input_tokens?: number; - total_cached_tokens?: number; - total_output_tokens?: number; - total_thought_tokens?: number; - total_tokens?: number; -} - -interface InteractionResource { - id?: string; - status?: string; - usage?: InteractionUsage; -} - -interface InteractionStreamMetadata { - total_usage?: InteractionUsage; -} - -interface InteractionSseEvent { - event_type?: string; - index?: number; - step?: InteractionStep; - delta?: - | InteractionContent - | InteractionFunctionCallStep - | InteractionThoughtStep - | InteractionThoughtSummaryDelta - | InteractionThoughtSignatureDelta - | InteractionArgumentsDelta; - interaction?: InteractionResource; - interaction_id?: string; - status?: string; - metadata?: InteractionStreamMetadata; - error?: { message?: string; code?: string | number }; -} - -function emptyUsage(): Usage { - return { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }; -} - -function getGoogleInteractionsState( - providerSessionState: Map | undefined, - create: boolean, -): GoogleInteractionsProviderSessionState | undefined { - if (!providerSessionState) return undefined; - const existing = providerSessionState.get(GOOGLE_INTERACTIONS_STATE_KEY) as - | GoogleInteractionsProviderSessionState - | undefined; - if (existing || !create) return existing; - const state: GoogleInteractionsProviderSessionState = { close: () => {} }; - providerSessionState.set(GOOGLE_INTERACTIONS_STATE_KEY, state); - return state; -} - -function interactionContentFromText(text: string): InteractionContent[] { - return text.length === 0 ? [] : [{ type: "text", text }]; -} - -function interactionContentFromParts(parts: readonly (TextContent | ImageContent)[]): InteractionContent[] { - const content: InteractionContent[] = []; - for (const part of parts) { - if (part.type === "text") { - if (part.text.length > 0) content.push({ type: "text", text: part.text }); - } else { - content.push({ type: "image", data: part.data, mime_type: part.mimeType }); - } - } - return content; -} - -function userInputStepFromMessage(message: Extract): InteractionUserInputStep { - const content = - typeof message.content === "string" - ? interactionContentFromText(message.content) - : interactionContentFromParts(message.content); - return { type: "user_input", content }; -} - -function functionResultStepFromMessage( - message: Extract, -): InteractionFunctionResultStep { - const result = interactionContentFromParts(message.content); - return { - type: "function_result", - name: message.toolName, - call_id: message.toolCallId, - result: result.length > 0 ? result : [{ type: "text", text: "" }], - ...(message.isError ? { is_error: true } : {}), - }; -} - -function appendAssistantInteractionSteps(message: AssistantMessage, steps: InteractionInputStep[]): void { - let modelContent: InteractionContent[] = []; - const flushModelContent = (): void => { - if (modelContent.length === 0) return; - steps.push({ type: "model_output", content: modelContent }); - modelContent = []; - }; - - for (const block of message.content) { - if (block.type === "text") { - if (block.text.length > 0) modelContent.push({ type: "text", text: block.text }); - } else if (block.type === "toolCall") { - flushModelContent(); - steps.push({ type: "function_call", id: block.id, name: block.name, arguments: block.arguments }); - } - } - flushModelContent(); -} - -function interactionMessagesAfterAnchor( - messages: readonly Message[], - anchorIndex: number | undefined, -): readonly Message[] { - return anchorIndex === undefined ? messages : messages.slice(anchorIndex + 1); -} - -function buildInteractionInput(context: Context, anchorIndex: number | undefined): InteractionInputStep[] { - const input: InteractionInputStep[] = []; - for (const message of interactionMessagesAfterAnchor(context.messages, anchorIndex)) { - if (message.role === "user" || message.role === "developer") { - const step = userInputStepFromMessage(message); - if (step.content.length > 0) input.push(step); - } else if (message.role === "toolResult") { - input.push(functionResultStepFromMessage(message)); - } else if (anchorIndex === undefined) { - appendAssistantInteractionSteps(message, input); - } - } - return input.length > 0 ? input : [{ type: "user_input", content: [{ type: "text", text: "" }] }]; -} - -function toInteractionThinkingLevel(level: GoogleThinkingLevel): InteractionThinkingLevel | undefined { - switch (level) { - case "MINIMAL": - return "minimal"; - case "LOW": - return "low"; - case "MEDIUM": - return "medium"; - case "HIGH": - return "high"; - case "THINKING_LEVEL_UNSPECIFIED": - return undefined; - } -} - -function buildInteractionGenerationConfig(options: GoogleOptions | undefined): InteractionGenerationConfig | undefined { - const config: InteractionGenerationConfig = {}; - if (options?.temperature !== undefined) config.temperature = options.temperature; - if (options?.topP !== undefined) config.top_p = options.topP; - if (options?.topK !== undefined) config.top_k = options.topK; - if (options?.minP !== undefined) config.min_p = options.minP; - if (options?.presencePenalty !== undefined) config.presence_penalty = options.presencePenalty; - if (options?.frequencyPenalty !== undefined) config.frequency_penalty = options.frequencyPenalty; - if (options?.repetitionPenalty !== undefined) config.repetition_penalty = options.repetitionPenalty; - if (options?.maxTokens !== undefined) config.max_output_tokens = options.maxTokens; - if (options?.thinking?.level !== undefined) { - const thinkingLevel = toInteractionThinkingLevel(options.thinking.level); - if (thinkingLevel !== undefined) config.thinking_level = thinkingLevel; - } else if (options?.thinking?.budgetTokens !== undefined) { - config.thinking_budget = options.thinking.budgetTokens; - } - return Object.keys(config).length > 0 ? config : undefined; -} - -function buildInteractionRequest( - model: GoogleInteractionsModel, - context: Context, - options: GoogleOptions | undefined, - anchor: InteractionAnchor, -): GoogleInteractionRequest { - const systemInstruction = normalizeSystemPrompts(context.systemPrompt).join("\n\n"); - const generationConfig = buildInteractionGenerationConfig(options); - return { - model: model.id, - input: buildInteractionInput(context, anchor.messageIndex), - stream: true, - ...(anchor.id !== undefined ? { previous_interaction_id: anchor.id } : {}), - ...(systemInstruction.length > 0 ? { system_instruction: systemInstruction } : {}), - ...(context.tools && context.tools.length > 0 ? { tools: convertTools(context.tools, model) } : {}), - ...(options?.storeInteraction !== undefined ? { store: options.storeInteraction } : {}), - ...(generationConfig !== undefined ? { generation_config: generationConfig } : {}), - ...(shouldSendServiceTier(options?.serviceTier, model.provider) ? { service_tier: options?.serviceTier } : {}), - }; -} - -function applyInteractionUsage( - model: GoogleInteractionsModel, - output: AssistantMessage, - usage: InteractionUsage, -): void { - const thinkingTokens = usage.total_thought_tokens ?? 0; - output.usage = { - input: (usage.total_input_tokens ?? 0) - (usage.total_cached_tokens ?? 0), - output: (usage.total_output_tokens ?? 0) + thinkingTokens, - cacheRead: usage.total_cached_tokens ?? 0, - cacheWrite: 0, - totalTokens: usage.total_tokens ?? 0, - ...(thinkingTokens > 0 ? { reasoningTokens: thinkingTokens } : {}), - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }; - calculateCost(model, output.usage); -} - -function parseInteractionFunctionCall(value: unknown): InteractionFunctionCallStep | undefined { - if (!value || typeof value !== "object" || Array.isArray(value)) return undefined; - const record = value as Record; - if (record.type !== "function_call") return undefined; - if (typeof record.id !== "string" || typeof record.name !== "string") return undefined; - const args = record.arguments; - return { - type: "function_call", - id: record.id, - name: record.name, - arguments: args && typeof args === "object" && !Array.isArray(args) ? { ...args } : {}, - }; -} - -function pendingToolCallFromStep(call: InteractionFunctionCallStep): PendingInteractionToolCall { - return { - id: call.id, - name: call.name, - argumentsText: "", - argumentsObject: call.arguments, - }; -} - -function parseInteractionArguments(text: string): Record { - if (text.trim().length === 0) return {}; - try { - const parsed: unknown = JSON.parse(text); - if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return { ...parsed }; - } catch { - return {}; - } - return {}; -} - -/** Provider-specific URL, headers, and fetch implementation for an Interactions request. */ -export interface GoogleInteractionsPlan { - url: string; - headers: Record; - fetch?: FetchImpl; -} - -/** - * Streams Gemini Interactions API model-mode responses for direct Google and Vertex providers. - * - * `fallback`, when supplied, is the legacy `:streamGenerateContent` stream factory. It runs - * transparently — forwarding its events into this stream — when the Interactions attempt fails - * before any content is emitted with a signal that the model/endpoint does not support - * Interactions (HTTP 404/400). Provide it only for auto-selected Interactions requests so an - * explicit `useInteractionsApi: true` still surfaces failures. - */ -export function streamGoogleInteractions(args: { - model: Model; - context: Context; - options: GoogleSharedStreamOptions | undefined; - api: T; - anchor: InteractionAnchor; - state: GoogleInteractionsProviderSessionState | undefined; - prepare: () => GoogleInteractionsPlan | Promise; - fallback?: () => AssistantMessageEventStream; -}): AssistantMessageEventStream { - const { model, context, options, anchor, state } = args; - const stream = new AssistantMessageEventStream(); - const output: AssistantMessage = { - role: "assistant", - content: [], - api: args.api, - provider: model.provider, - model: model.id, - usage: emptyUsage(), - stopReason: "stop", - timestamp: Date.now(), - }; - const storeInteraction = options?.storeInteraction !== false; - - void (async () => { - let started = false; - let sawTerminal = false; - let currentTextBlock: TextContent | undefined; - let currentThinkingBlock: Extract | undefined; - let pendingThinkingSignature: string | undefined; - const stepKinds = new Map(); - const pendingToolCalls = new Map(); - const ensureStarted = (): void => { - if (started) return; - stream.push({ type: "start", partial: output }); - started = true; - }; - const endOpenBlocks = (): void => { - if (currentTextBlock) { - stream.push({ - type: "text_end", - contentIndex: output.content.indexOf(currentTextBlock), - content: currentTextBlock.text, - partial: output, - }); - currentTextBlock = undefined; - } - if (currentThinkingBlock) { - stream.push({ - type: "thinking_end", - contentIndex: output.content.indexOf(currentThinkingBlock), - content: currentThinkingBlock.thinking, - partial: output, - }); - currentThinkingBlock = undefined; - } - }; - const emitText = (text: string): void => { - if (text.length === 0) return; - ensureStarted(); - if (!currentTextBlock) { - if (currentThinkingBlock) endOpenBlocks(); - currentTextBlock = { type: "text", text: "" }; - output.content.push(currentTextBlock); - stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output }); - } - currentTextBlock.text += text; - stream.push({ - type: "text_delta", - contentIndex: output.content.indexOf(currentTextBlock), - delta: text, - partial: output, - }); - }; - const applyThinkingSignature = (signature: string | undefined): void => { - if (!signature) return; - if (currentThinkingBlock) { - currentThinkingBlock.thinkingSignature = signature; - } else { - pendingThinkingSignature = signature; - } - }; - const emitThinking = (text: string): void => { - if (text.length === 0) return; - ensureStarted(); - if (!currentThinkingBlock) { - if (currentTextBlock) endOpenBlocks(); - currentThinkingBlock = { type: "thinking", thinking: "", thinkingSignature: pendingThinkingSignature }; - pendingThinkingSignature = undefined; - output.content.push(currentThinkingBlock); - stream.push({ type: "thinking_start", contentIndex: output.content.length - 1, partial: output }); - } - currentThinkingBlock.thinking += text; - stream.push({ - type: "thinking_delta", - contentIndex: output.content.indexOf(currentThinkingBlock), - delta: text, - partial: output, - }); - }; - const emitToolCall = (call: InteractionFunctionCallStep): void => { - ensureStarted(); - endOpenBlocks(); - const toolCall: ToolCall = { - type: "toolCall", - id: call.id, - name: call.name, - arguments: call.arguments, - }; - output.content.push(toolCall); - const contentIndex = output.content.length - 1; - stream.push({ type: "toolcall_start", contentIndex, partial: output }); - stream.push({ - type: "toolcall_delta", - contentIndex, - delta: JSON.stringify(toolCall.arguments), - partial: output, - }); - stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); - }; - const emitPendingToolCall = (pending: PendingInteractionToolCall): void => { - emitToolCall({ - type: "function_call", - id: pending.id, - name: pending.name, - arguments: - pending.argumentsText.length > 0 - ? parseInteractionArguments(pending.argumentsText) - : pending.argumentsObject, - }); - }; - - let prepared = false; - try { - const plan = await args.prepare(); - prepared = true; - let requestBody: unknown = buildInteractionRequest(model, context, options, anchor); - const replacement = await options?.onPayload?.(requestBody, model); - if (replacement !== undefined) requestBody = replacement; - const response = await fetchWithRetry(() => plan.url, { - method: "POST", - headers: { - ...plan.headers, - "Content-Type": "application/json", - Accept: "text/event-stream", - }, - body: JSON.stringify(requestBody), - signal: options?.signal, - fetch: plan.fetch, - }); - if (!response.ok) { - const errorText = await response.text().catch(() => ""); - throw new AIError.GoogleApiError( - `Google Interactions API error (${response.status}): ${errorText}`, - response.status, - { - headers: response.headers, - }, - ); - } - if (!response.body) { - throw new AIError.ProviderResponseError("Google Interactions API returned an empty response body", { - provider: model.provider, - kind: "empty-body", - }); - } - for await (const event of readSseJson(response.body, options?.signal, sse => - options?.onSseEvent?.({ event: sse.event, data: sse.data, raw: [...sse.raw] }, model), - )) { - if (event.error) { - throw new AIError.ProviderResponseError(event.error.message ?? "Google Interactions API stream error", { - provider: model.provider, - kind: "runtime", - }); - } - if (event.metadata?.total_usage) applyInteractionUsage(model, output, event.metadata.total_usage); - if (event.event_type === "interaction.created") { - if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; - } else if (event.event_type === "step.start" && event.index !== undefined && event.step) { - stepKinds.set(event.index, event.step.type); - const call = parseInteractionFunctionCall(event.step); - if (call) { - pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); - } else if (event.step.type === "thought") { - applyThinkingSignature(event.step.signature); - for (const item of event.step.summary ?? []) { - if (item.type === "text") emitThinking(item.text); - } - } - } else if (event.event_type === "step.delta" && event.index !== undefined && event.delta) { - const call = parseInteractionFunctionCall(event.delta); - if (call) { - pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); - } else if (event.delta.type === "text") { - if (stepKinds.get(event.index) === "thought") emitThinking(event.delta.text); - else emitText(event.delta.text); - } else if (event.delta.type === "thought_summary") { - if (event.delta.content?.type === "text") emitThinking(event.delta.content.text); - } else if (event.delta.type === "thought_signature") { - applyThinkingSignature(event.delta.signature); - } else if (event.delta.type === "arguments_delta") { - const pending = pendingToolCalls.get(event.index); - if (pending && event.delta.arguments) pending.argumentsText += event.delta.arguments; - } - } else if (event.event_type === "step.stop" && event.index !== undefined) { - const stepKind = stepKinds.get(event.index); - const pending = pendingToolCalls.get(event.index); - if (pending) { - emitPendingToolCall(pending); - pendingToolCalls.delete(event.index); - } else { - endOpenBlocks(); - if (stepKind === "thought") pendingThinkingSignature = undefined; - } - stepKinds.delete(event.index); - } else if (event.event_type === "interaction.completed" || event.event_type === "interaction.complete") { - if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; - if (event.interaction?.usage) applyInteractionUsage(model, output, event.interaction.usage); - for (const pending of pendingToolCalls.values()) emitPendingToolCall(pending); - pendingToolCalls.clear(); - endOpenBlocks(); - output.stopReason = - event.interaction?.status === "requires_action" || - output.content.some(block => block.type === "toolCall") - ? "toolUse" - : "stop"; - if (storeInteraction && state) state.lastInteractionId = output.responseId; - sawTerminal = true; - ensureStarted(); - stream.push({ type: "done", reason: output.stopReason, message: output }); - } - } - if (!sawTerminal) { - throw new AIError.ProviderResponseError("Google Interactions API stream ended without a terminal event", { - provider: model.provider, - kind: "incomplete-stream", - }); - } - } catch (error) { - // Auto-selected Interactions degrades to `:streamGenerateContent` when no content has - // streamed yet and the failure means Interactions can't serve this request — the bearer - // credential couldn't be resolved (prepare threw) or the model/endpoint rejected it - // (HTTP 404/400). Provider 401/403/429/5xx still surface. Mirrors the OpenAI Responses - // `previous_response_id` fallback. - const unsupported = error instanceof AIError.GoogleApiError && (error.status === 404 || error.status === 400); - if (!started && args.fallback && !options?.signal?.aborted && (!prepared || unsupported)) { - for await (const event of args.fallback()) stream.push(event); - return; - } - output.stopReason = options?.signal?.aborted ? "aborted" : "error"; - output.errorMessage = error instanceof Error ? error.message : String(error); - stream.push({ type: "error", reason: output.stopReason, error: output }); - } - })(); - - return stream; -} - -function findAssistantInteractionAnchor( - context: Context, - interactionId: string, - provider: string, -): InteractionAnchor | undefined { - for (let index = context.messages.length - 1; index >= 0; index -= 1) { - const message = context.messages[index]; - if (message?.role === "assistant" && message.provider === provider && message.responseId === interactionId) { - return { id: interactionId, messageIndex: index }; - } - } - return undefined; -} - -function latestAssistantInteractionAnchor(context: Context, provider: string): InteractionAnchor | undefined { - for (let index = context.messages.length - 1; index >= 0; index -= 1) { - const message = context.messages[index]; - if (message?.role === "assistant" && message.provider === provider && message.responseId) { - return { id: message.responseId, messageIndex: index }; - } - } - return undefined; -} - -function resolveInteractionAnchor( - context: Context, - explicitPreviousInteractionId: string | undefined, - state: GoogleInteractionsProviderSessionState | undefined, - provider: string, -): InteractionAnchor { - if (explicitPreviousInteractionId !== undefined) { - return ( - findAssistantInteractionAnchor(context, explicitPreviousInteractionId, provider) ?? { - id: explicitPreviousInteractionId, - } - ); - } - const lineageAnchor = latestAssistantInteractionAnchor(context, provider); - if (lineageAnchor) return lineageAnchor; - if (state?.lastInteractionId) - return findAssistantInteractionAnchor(context, state.lastInteractionId, provider) ?? {}; - return {}; -} - -/** - * Whether a model is served by the Gemini Interactions API. Interactions is a Gemini 3-era - * transport, so the catalog subset that supports it is Gemini 3.0+. Older Gemini and non-Gemini - * ids keep `:streamGenerateContent`, which covers the full catalog. - */ -export function modelSupportsInteractions(model: Pick): boolean { - const parsed = parseGeminiModel(model.id); - return parsed !== null && parsed.version.major >= 3; -} - -/** - * Resolves whether a Google provider call should use Interactions and which lineage anchor to send. - * - * Precedence: explicit `useInteractionsApi: false` always wins (force generateContent); otherwise - * Interactions engages when explicitly requested, when continuing a stored interaction - * (`previousInteractionId`/assistant lineage/session state), or when `autoEligible` (the - * zero-config default for the capable model subset on the official endpoint). `auto` flags the - * last case for the caller — it is the only mode that wires up the generateContent fallback. - */ -export function resolveInteractionDispatch(args: { - context: Context; - options: GoogleSharedStreamOptions | undefined; - provider: string; - autoEligible: boolean; -}): { - useInteractions: boolean; - auto: boolean; - anchor: InteractionAnchor; - state: GoogleInteractionsProviderSessionState | undefined; -} { - const explicitPreviousInteractionId = args.options?.previousInteractionId; - if (args.options?.storeInteraction === false && explicitPreviousInteractionId !== undefined) { - throw new AIError.ConfigurationError( - "Google Interactions API cannot combine storeInteraction:false with previousInteractionId.", - ); - } - const explicitOptOut = args.options?.useInteractionsApi === false; - const explicitOptIn = args.options?.useInteractionsApi === true || explicitPreviousInteractionId !== undefined; - const storageEnabled = args.options?.storeInteraction !== false; - const existingState = storageEnabled - ? getGoogleInteractionsState(args.options?.providerSessionState, false) - : undefined; - const anchor = explicitOptOut - ? {} - : resolveInteractionAnchor(args.context, explicitPreviousInteractionId, existingState, args.provider); - const useInteractions = !explicitOptOut && (explicitOptIn || anchor.id !== undefined || args.autoEligible); - const auto = useInteractions && !explicitOptIn; - const interactionState = - useInteractions && storageEnabled - ? getGoogleInteractionsState( - args.options?.providerSessionState, - args.options?.providerSessionState !== undefined, - ) - : undefined; - return { useInteractions, auto, anchor, state: interactionState }; -} diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index 28c805b41..2559af6d4 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -79,19 +79,6 @@ export interface GoogleSharedStreamOptions extends StreamOptions { hideThinkingSummary?: boolean; /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ serviceTier?: ServiceTier; - /** - * Continues a Gemini Interactions API conversation from a stored interaction. - * When set on the direct Google provider, the request uses `/interactions` - * with `previous_interaction_id` instead of the legacy generateContent stream. - */ - previousInteractionId?: string; - /** - * Uses the Gemini Interactions API for direct Google requests, storing the - * returned interaction id on the assistant response for follow-up turns. - */ - useInteractionsApi?: boolean; - /** Overrides Interactions API request storage; default is the API default (`true`). */ - storeInteraction?: boolean; } /** diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 66b1868b9..647c2c8d5 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -2,13 +2,7 @@ import { $env } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { getVertexAccessToken, hasVertexBearerCredentialsHint } from "./google-auth"; -import { - type GoogleInteractionsPlan, - modelSupportsInteractions, - resolveInteractionDispatch, - streamGoogleInteractions, -} from "./google-interactions"; +import { getVertexAccessToken } from "./google-auth"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -22,126 +16,86 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions { } const API_VERSION = "v1"; -const INTERACTIONS_API_VERSION = "v1beta1"; -const INTERACTIONS_API_REVISION = "2026-05-20"; export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( model: Model<"google-vertex">, context: Context, options?: GoogleVertexOptions, ): AssistantMessageEventStream => { - const runGenerateContent = (): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-vertex", - retainTextSignature: true, - prepare: async (): Promise => { - const apiKey = resolveApiKey(options); - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - params.config ||= {}; - if (!params.config.safetySettings) { - params.config.safetySettings = [ - { - category: "HARM_CATEGORY_HATE_SPEECH", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_DANGEROUS_CONTENT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", - threshold: "OFF", - }, - { - category: "HARM_CATEGORY_HARASSMENT", - threshold: "OFF", - }, - ]; - } - const baseHeaders: Record = { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; - // Vertex AI ignores a `serviceTier` request-body field (unlike the direct - // Gemini API); priority must travel as a request header. Only `priority` - // has a documented Vertex request control — `flex` has none, so it's a no-op. - if (options?.serviceTier === "priority") { - baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; - } - - if (apiKey) { - // Explicit `location` is a deliberate residency choice: honor it and let - // a 404 surface. An ambient env-derived region falls back to the global - // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working - // global-only request. - const explicitLocation = options?.location; - const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; - const host = resolveEndpointHost(location); - const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; - const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; - return { - params, - url: `https://${host}/${path}`, - fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, - headers: { - ...baseHeaders, - "x-goog-api-key": apiKey, - }, - fetch: options?.fetch, - }; - } - - const project = resolveProject(options); - const location = resolveLocation(options); - const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); - const host = resolveEndpointHost(location); - const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; - return { - params, - url, - headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, - fetch: options?.fetch, - }; - }, - }); - - // Default Gemini 3+ onto Interactions whenever a bearer credential source exists (ADC file, - // `GOOGLE_APPLICATION_CREDENTIALS`, or an explicit access-token env). Interactions needs bearer - // auth, so express API-key-only setups stay on generateContent — and an express key, when - // present, still serves the generateContent fallback. Interactions always targets the official - // global `aiplatform` host; the fallback also recovers ids the endpoint rejects. - const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ - context, - options, - provider: model.provider, - autoEligible: modelSupportsInteractions(model) && hasVertexBearerCredentialsHint(), - }); - if (!useInteractions) return runGenerateContent(); - - return streamGoogleInteractions({ + return streamGoogleGenAI({ model, - context, options, api: "google-vertex", - anchor, - state, - prepare: async (): Promise => { + retainTextSignature: true, + prepare: async (): Promise => { + const apiKey = resolveApiKey(options); + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + params.config ||= {}; + if (!params.config.safetySettings) { + params.config.safetySettings = [ + { + category: "HARM_CATEGORY_HATE_SPEECH", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_DANGEROUS_CONTENT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_HARASSMENT", + threshold: "OFF", + }, + ]; + } + const baseHeaders: Record = { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + // Vertex AI ignores a `serviceTier` request-body field (unlike the direct + // Gemini API); priority must travel as a request header. Only `priority` + // has a documented Vertex request control — `flex` has none, so it's a no-op. + if (options?.serviceTier === "priority") { + baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; + } + + if (apiKey) { + // Explicit `location` is a deliberate residency choice: honor it and let + // a 404 surface. An ambient env-derived region falls back to the global + // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working + // global-only request. + const explicitLocation = options?.location; + const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + const host = resolveEndpointHost(location); + const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; + return { + params, + url: `https://${host}/${path}`, + fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, + headers: { + ...baseHeaders, + "x-goog-api-key": apiKey, + }, + fetch: options?.fetch, + }; + } + const project = resolveProject(options); + const location = resolveLocation(options); const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); + const host = resolveEndpointHost(location); + const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { - url: `https://aiplatform.googleapis.com/${INTERACTIONS_API_VERSION}/projects/${project}/locations/global/interactions`, - headers: { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - Authorization: `Bearer ${accessToken}`, - "Api-Revision": INTERACTIONS_API_REVISION, - }, + params, + url, + headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, fetch: options?.fetch, }; }, - fallback: auto ? runGenerateContent : undefined, }); }; diff --git a/packages/ai/src/providers/google.ts b/packages/ai/src/providers/google.ts index 3bd8ed7ff..f9d38a1f6 100644 --- a/packages/ai/src/providers/google.ts +++ b/packages/ai/src/providers/google.ts @@ -2,7 +2,6 @@ import * as AIError from "../error"; import { getEnvApiKey } from "../stream"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { modelSupportsInteractions, resolveInteractionDispatch, streamGoogleInteractions } from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -27,61 +26,22 @@ export const streamGoogle: StreamFunction<"google-generative-ai"> = ( ); } - const runGenerateContent = (): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-generative-ai", - prepare: (): GoogleGenAIRequestPlan => { - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - // `model.baseUrl` already includes the API version segment when set (mirrors the - // `apiVersion: ""` reset that the SDK relied on for custom base URLs). - const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; - const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; - const headers: Record = { - "x-goog-api-key": apiKey, - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; - return { params, url, headers, fetch: options?.fetch }; - }, - }); - - // Default Gemini 3+ on the official endpoint onto Interactions (custom proxy base URLs keep - // generateContent, which serves the full catalog). The fallback recovers ids the endpoint rejects. - const trimmedBase = model.baseUrl?.trim(); - let officialEndpoint = !trimmedBase; - if (trimmedBase) { - try { - officialEndpoint = new URL(trimmedBase).hostname === "generativelanguage.googleapis.com"; - } catch { - officialEndpoint = false; - } - } - const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ - context, - options, - provider: model.provider, - autoEligible: officialEndpoint && modelSupportsInteractions(model), - }); - if (!useInteractions) return runGenerateContent(); - - return streamGoogleInteractions({ + return streamGoogleGenAI({ model, - context, options, api: "google-generative-ai", - anchor, - state, - prepare: () => ({ - url: `${trimmedBase || DEFAULT_GENERATIVE_LANGUAGE_BASE}/interactions`, - headers: { + prepare: (): GoogleGenAIRequestPlan => { + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + // `model.baseUrl` already includes the API version segment when set (mirrors the + // `apiVersion: ""` reset that the SDK relied on for custom base URLs). + const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; + const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; + const headers: Record = { "x-goog-api-key": apiKey, ...(model.headers ?? {}), ...(options?.headers ?? {}), - }, - fetch: options?.fetch, - }), - fallback: auto ? runGenerateContent : undefined, + }; + return { params, url, headers, fetch: options?.fetch }; + }, }); }; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 7c66e6c9f..1d1dbd45b 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1432,9 +1432,6 @@ function mapOptionsForApi( streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs, streamIdleTimeoutMs: options?.streamIdleTimeoutMs, providerSessionState: options?.providerSessionState, - useInteractionsApi: options?.useInteractionsApi, - storeInteraction: options?.storeInteraction, - previousInteractionId: options?.previousInteractionId, maxInFlightRequests: options?.maxInFlightRequests, onPayload: options?.onPayload, onResponse: options?.onResponse, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 5b568de29..d5903e5dd 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -424,22 +424,6 @@ export interface StreamOptions { providerSessionState?: Map; /** Canonical Codex compaction classification; ignored by other providers. */ codexCompaction?: CodexCompactionRequestContext; - /** - * Force Gemini model-mode Interactions API transport for providers that support it. - * When unset, those providers may still use Interactions to continue known - * server-side conversation lineage via `previousInteractionId` or stored state. - */ - useInteractionsApi?: boolean; - /** - * Whether supported Interactions transports should store server-side conversation - * state and return response ids for follow-up turns. Defaults to true. - */ - storeInteraction?: boolean; - /** - * Explicit Interactions response id to continue. Mutually exclusive with - * `storeInteraction: false` because the follow-up itself must be storable. - */ - previousInteractionId?: string; /** * Optional per-provider concurrent request cap for LLM stream calls. Keys are * provider ids (`model.provider`); positive numeric values cap in-flight diff --git a/packages/ai/test/google-empty-response-retry.test.ts b/packages/ai/test/google-empty-response-retry.test.ts index f45464685..2e66440c1 100644 --- a/packages/ai/test/google-empty-response-retry.test.ts +++ b/packages/ai/test/google-empty-response-retry.test.ts @@ -89,8 +89,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return calls === 1 ? sse(genaiChunk("")) : sse(genaiChunk("Hello!")); }; - // Pin the generateContent transport: gemini-3 ids now auto-route to Interactions by default. - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); const { events, starts } = await drain(stream); const result = await stream.result(); @@ -108,7 +107,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return sse(genaiChunk("")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); const result = await stream.result(); expect(calls).toBe(3); // MAX_EMPTY_STREAM_RETRIES (2) + 1 initial attempt @@ -138,7 +137,6 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, - useInteractionsApi: false, }); const { events } = await drain(stream); const result = await stream.result(); @@ -196,7 +194,6 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, - useInteractionsApi: false, }); const result = await stream.result(); diff --git a/packages/ai/test/google-interactions.test.ts b/packages/ai/test/google-interactions.test.ts deleted file mode 100644 index 136fe2d95..000000000 --- a/packages/ai/test/google-interactions.test.ts +++ /dev/null @@ -1,511 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; -import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; -import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; -import { streamSimple } from "@oh-my-pi/pi-ai/stream"; -import type { AssistantMessage, Context, FetchImpl, Model, Tool, Usage } from "@oh-my-pi/pi-ai/types"; -import { buildModel } from "@oh-my-pi/pi-catalog/build"; - -function googleModel(baseUrl = "https://generativelanguage.googleapis.com/v1beta"): Model<"google-generative-ai"> { - return buildModel({ - id: "gemini-3.5-flash", - name: "Gemini 3.5 Flash", - api: "google-generative-ai", - provider: "google", - baseUrl, - reasoning: true, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1_000_000, - maxTokens: 8_192, - }); -} - -function vertexModel(id = "gemini-3.5-flash"): Model<"google-vertex"> { - return buildModel({ - id, - name: id, - api: "google-vertex", - provider: "google-vertex", - baseUrl: "", - reasoning: true, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1_000_000, - maxTokens: 8_192, - }); -} - -function sseResponse(events: readonly unknown[]): Response { - const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; - return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } }); -} - -const weatherTool: Tool = { - name: "get_weather", - description: "Get weather", - parameters: { - type: "object", - properties: { city: { type: "string" } }, - required: ["city"], - additionalProperties: false, - }, -}; - -describe("Google Interactions API", () => { - it("chains tool results with previous_interaction_id from the prior assistant response", async () => { - const model = googleModel(); - const requestBodies: unknown[] = []; - let calls = 0; - const fetchMock: FetchImpl = async (_input, init) => { - requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); - calls += 1; - if (calls === 1) { - return sseResponse([ - { - event_type: "interaction.created", - interaction: { id: "int_1", status: "in_progress" }, - }, - { event_type: "step.start", index: 0, step: { type: "thought" } }, - { - event_type: "step.delta", - index: 0, - delta: { type: "thought_signature", signature: "thought_sig_1" }, - }, - { - event_type: "step.delta", - index: 0, - delta: { type: "thought_summary", content: { type: "text", text: "Checking weather.\n" } }, - }, - { event_type: "step.stop", index: 0 }, - { - event_type: "step.start", - index: 1, - step: { - type: "function_call", - id: "call_weather", - name: "get_weather", - arguments: {}, - }, - }, - { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: '{"city":"Bos' } }, - { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: 'ton"}' } }, - { event_type: "step.stop", index: 1 }, - { - event_type: "interaction.completed", - interaction: { - id: "int_1", - status: "requires_action", - usage: { total_input_tokens: 10, total_output_tokens: 2, total_tokens: 12 }, - }, - }, - ]); - } - return sseResponse([ - { event_type: "interaction.created", interaction: { id: "int_2", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text: "Sunny." } }, - { event_type: "step.stop", index: 0 }, - { - event_type: "interaction.completed", - interaction: { - id: "int_2", - status: "completed", - usage: { total_input_tokens: 3, total_output_tokens: 1, total_tokens: 4 }, - }, - }, - ]); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const firstContext: Context = { - systemPrompt: ["Use concise weather reports."], - messages: [{ role: "user", content: "Need weather", timestamp: 1 }], - tools: [weatherTool], - }; - const first = await streamGoogle(model, firstContext, { - apiKey: "test-key", - fetch: fetchMock, - useInteractionsApi: true, - thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, - }).result(); - - expect(first.responseId).toBe("int_1"); - expect(first.stopReason).toBe("toolUse"); - expect(first.content).toEqual([ - { type: "thinking", thinking: "Checking weather.\n", thinkingSignature: "thought_sig_1" }, - { type: "toolCall", id: "call_weather", name: "get_weather", arguments: { city: "Boston" } }, - ]); - expect(requestBodies[0]).toMatchObject({ - model: "gemini-3.5-flash", - stream: true, - input: [{ type: "user_input", content: [{ type: "text", text: "Need weather" }] }], - system_instruction: "Use concise weather reports.", - tools: [{ functionDeclarations: [{ name: "get_weather" }] }], - generation_config: { thinking_level: "high" }, - }); - expect(requestBodies[0]).not.toHaveProperty("previous_interaction_id"); - expect(JSON.stringify(requestBodies[0])).not.toContain("thinking_budget"); - - const secondContext: Context = { - messages: [ - { role: "user", content: "Need weather", timestamp: 1 }, - first, - { - role: "toolResult", - toolCallId: "call_weather", - toolName: "get_weather", - content: [{ type: "text", text: "72F and sunny" }], - isError: false, - timestamp: 2, - }, - ], - tools: [weatherTool], - systemPrompt: ["Use concise weather reports."], - }; - const second = await streamGoogle(model, secondContext, { - apiKey: "test-key", - fetch: fetchMock, - thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, - }).result(); - - expect(second.responseId).toBe("int_2"); - expect(second.content).toEqual([{ type: "text", text: "Sunny." }]); - expect(requestBodies[1]).toMatchObject({ - previous_interaction_id: "int_1", - input: [ - { - type: "function_result", - name: "get_weather", - call_id: "call_weather", - result: [{ type: "text", text: "72F and sunny" }], - }, - ], - tools: [{ functionDeclarations: [{ name: "get_weather" }] }], - system_instruction: "Use concise weather reports.", - generation_config: { thinking_level: "high" }, - }); - expect(JSON.stringify(requestBodies[1])).not.toContain("Need weather"); - expect(JSON.stringify(requestBodies[1])).not.toContain("thinking_budget"); - }); - - it("does not expose or reuse interaction ids when storage is disabled", async () => { - const model = googleModel(); - const requestBodies: unknown[] = []; - const fetchMock: FetchImpl = async (_input, init) => { - requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); - return sseResponse([ - { event_type: "interaction.created", interaction: { id: "unstored_int", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text: "Done." } }, - { event_type: "step.stop", index: 0 }, - { event_type: "interaction.completed", interaction: { id: "unstored_int", status: "completed" } }, - ]); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const result = await streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { - apiKey: "test-key", - fetch: fetchMock, - useInteractionsApi: true, - storeInteraction: false, - }, - ).result(); - - expect(result.responseId).toBeUndefined(); - expect(requestBodies[0]).toMatchObject({ store: false }); - expect(() => - streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { - apiKey: "test-key", - fetch: fetchMock, - storeInteraction: false, - previousInteractionId: "unstored_int", - }, - ), - ).toThrow(/storeInteraction:false/); - }); - - it("reads thought payload from step.start without leaking a prior signature", async () => { - const model = googleModel(); - const fetchMock: FetchImpl = async () => - sseResponse([ - { event_type: "interaction.created", interaction: { id: "int_3", status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "thought", signature: "stale_sig" } }, - { event_type: "step.stop", index: 0 }, - { - event_type: "step.start", - index: 1, - step: { type: "thought", summary: [{ type: "text", text: "Fresh plan.\n" }] }, - }, - { event_type: "step.stop", index: 1 }, - { event_type: "step.start", index: 2, step: { type: "model_output" } }, - { event_type: "step.delta", index: 2, delta: { type: "text", text: "Answer." } }, - { event_type: "step.stop", index: 2 }, - { event_type: "interaction.completed", interaction: { id: "int_3", status: "completed" } }, - ]); - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - - const result = await streamGoogle( - model, - { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, - { apiKey: "test-key", fetch: fetchMock, useInteractionsApi: true }, - ).result(); - - expect(result.content).toEqual([ - { type: "thinking", thinking: "Fresh plan.\n" }, - { type: "text", text: "Answer." }, - ]); - }); -}); - -function genaiSse(text: string): Response { - return new Response( - `data: ${JSON.stringify({ - candidates: [{ content: { parts: [{ text }] }, finishReason: "STOP" }], - usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, - })}\n\n`, - { status: 200, headers: { "content-type": "text/event-stream" } }, - ); -} - -function interactionsTextSse( - id: string, - text: string, - terminal: "interaction.completed" | "interaction.complete" = "interaction.completed", -): Response { - return sseResponse([ - { event_type: "interaction.created", interaction: { id, status: "in_progress" } }, - { event_type: "step.start", index: 0, step: { type: "model_output" } }, - { event_type: "step.delta", index: 0, delta: { type: "text", text } }, - { event_type: "step.stop", index: 0 }, - { - event_type: terminal, - interaction: { - id, - status: "completed", - usage: { total_input_tokens: 10, total_output_tokens: 5, total_tokens: 15 }, - }, - }, - ]); -} - -interface CapturedCall { - url: string; - method: string; - headers: Headers; - body: unknown; -} - -function captureFetch(handler: (url: string) => Response): { fetch: FetchImpl; calls: CapturedCall[] } { - const calls: CapturedCall[] = []; - const fetchMock: FetchImpl = async (input, init) => { - const url = input instanceof Request ? input.url : String(input); - calls.push({ - url, - method: String(init?.method ?? "GET"), - headers: new Headers(init?.headers), - body: init?.body ? JSON.parse(String(init.body)) : undefined, - }); - return handler(url); - }; - Object.assign(fetchMock, { preconnect: fetch.preconnect }); - return { fetch: fetchMock, calls }; -} - -const ZERO_USAGE: Usage = { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, -}; - -function assistantWithResponse( - api: "google-vertex" | "google-generative-ai", - provider: string, - responseId: string, -): AssistantMessage { - return { - role: "assistant", - api, - provider, - model: "gemini-3.5-flash", - content: [{ type: "text", text: "prev" }], - usage: ZERO_USAGE, - stopReason: "stop", - timestamp: 2, - responseId, - }; -} - -const userTurn: Context = { messages: [{ role: "user", content: "Hi there", timestamp: 1 }] }; - -describe("Google Interactions API — zero-config default + fallback", () => { - let savedToken: string | undefined; - beforeEach(() => { - // A bearer source makes the Vertex auto-gate fire deterministically on any machine, and - // `getVertexAccessToken` returns it directly (no OAuth/metadata round-trip in tests). - savedToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; - Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = "test-bearer"; - }); - afterEach(() => { - if (savedToken === undefined) delete Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; - else Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = savedToken; - __resetVertexTokenCache(); - }); - - it("auto-routes a capable Vertex model to Interactions under bearer auth", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("vint_1", "Hi")); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls).toHaveLength(1); - expect(calls[0].url).toBe("https://aiplatform.googleapis.com/v1beta1/projects/p/locations/global/interactions"); - expect(calls[0].method).toBe("POST"); - expect(calls[0].headers.get("Api-Revision")).toBe("2026-05-20"); - expect(calls[0].headers.get("Authorization")).toBe("Bearer test-bearer"); - expect(calls[0].body).toMatchObject({ - model: "gemini-3.5-flash", - stream: true, - input: [{ type: "user_input", content: [{ type: "text", text: "Hi there" }] }], - }); - expect(calls[0].body).not.toHaveProperty("contents"); - expect(calls[0].body).not.toHaveProperty("agent"); - expect(calls[0].body).not.toHaveProperty("environment"); - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "Hi" }]); - expect(result.responseId).toBe("vint_1"); - expect(result.usage.totalTokens).toBe(15); - }); - - it("keeps an older (sub-3) Vertex model on generateContent", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogleVertex(vertexModel("gemini-2.5-flash"), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("honors useInteractionsApi:false on a capable Vertex model", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - useInteractionsApi: false, - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("falls back to generateContent when auto Interactions is unsupported (404)", async () => { - const { fetch, calls } = captureFetch(url => - url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("recovered"), - ); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - location: "us", - fetch, - }).result(); - - expect(calls[0].url).toContain("/interactions"); - expect(calls[1].url).toContain(":streamGenerateContent"); - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "recovered" }]); - }); - - it("surfaces the error (no fallback) when explicit Interactions is unsupported", async () => { - const { fetch, calls } = captureFetch(url => - url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("unexpected"), - ); - const result = await streamGoogleVertex(vertexModel(), userTurn, { - project: "p", - useInteractionsApi: true, - fetch, - }).result(); - - expect(result.stopReason).toBe("error"); - expect(calls.some(c => c.url.includes(":streamGenerateContent"))).toBe(false); - }); - - it("accepts interaction.complete as a terminal-event alias", async () => { - const { fetch } = captureFetch(() => interactionsTextSse("vint_2", "Done", "interaction.complete")); - const result = await streamGoogleVertex(vertexModel(), userTurn, { project: "p", fetch }).result(); - - expect(result.stopReason).toBe("stop"); - expect(result.content).toEqual([{ type: "text", text: "Done" }]); - }); - - it("sends previous_interaction_id only for same-provider assistant lineage", async () => { - const sameProvider = captureFetch(() => interactionsTextSse("vint_3", "ok")); - await streamGoogleVertex( - vertexModel(), - { - messages: [ - { role: "user", content: "a", timestamp: 1 }, - assistantWithResponse("google-vertex", "google-vertex", "vint_prev"), - { role: "user", content: "b", timestamp: 3 }, - ], - }, - { project: "p", fetch: sameProvider.fetch }, - ).result(); - expect(sameProvider.calls[0].body).toMatchObject({ previous_interaction_id: "vint_prev" }); - - const wrongProvider = captureFetch(() => interactionsTextSse("vint_4", "ok")); - await streamGoogleVertex( - vertexModel(), - { - messages: [ - { role: "user", content: "a", timestamp: 1 }, - assistantWithResponse("google-generative-ai", "google", "gint_prev"), - { role: "user", content: "b", timestamp: 3 }, - ], - }, - { project: "p", fetch: wrongProvider.fetch }, - ).result(); - expect(wrongProvider.calls[0].body).not.toHaveProperty("previous_interaction_id"); - }); - - it("auto-routes a capable direct Google model on the official endpoint to Interactions", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("gint_1", "Hi")); - const result = await streamGoogle(googleModel(), userTurn, { apiKey: "k", fetch }).result(); - - expect(calls[0].url).toBe("https://generativelanguage.googleapis.com/v1beta/interactions"); - expect(calls[0].headers.get("x-goog-api-key")).toBe("k"); - expect(result.responseId).toBe("gint_1"); - }); - - it("keeps a custom-baseUrl direct Google model on generateContent", async () => { - const { fetch, calls } = captureFetch(() => genaiSse("ok")); - await streamGoogle(googleModel("https://proxy.example.com/v1beta"), userTurn, { - apiKey: "k", - fetch, - }).result(); - - expect(calls[0].url).toContain(":streamGenerateContent"); - expect(calls[0].url.startsWith("https://proxy.example.com/")).toBe(true); - expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); - }); - - it("threads the auto-default through streamSimple for a capable Google model", async () => { - const { fetch, calls } = captureFetch(() => interactionsTextSse("sint_1", "Hi")); - await streamSimple(googleModel(), userTurn, { apiKey: "k", fetch }).result(); - - expect(calls.some(c => c.url.includes("/interactions"))).toBe(true); - }); -}); diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts index 74066c206..4122dcd92 100644 --- a/packages/ai/test/google-service-tier.test.ts +++ b/packages/ai/test/google-service-tier.test.ts @@ -75,9 +75,7 @@ const vertexModel: Model<"google-vertex"> = buildModel({ describe("Google service tier wire encoding", () => { it("Gemini API sends the tier in the request body, not a header", async () => { const { fetch, captured } = capturingFetch(); - await drain( - streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch, useInteractionsApi: false }), - ); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch })); const { headers, body } = captured(); expect(body.serviceTier).toBe("priority"); expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); @@ -89,7 +87,6 @@ describe("Google service tier wire encoding", () => { streamGoogle(geminiModel, context, { apiKey: "k", fetch, - useInteractionsApi: false, thinking: { enabled: true, level: "HIGH" }, hideThinkingSummary: true, }), @@ -119,7 +116,7 @@ describe("Google service tier wire encoding", () => { it("omits the tier entirely when unset", async () => { const { fetch, captured } = capturingFetch(); - await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch, useInteractionsApi: false })); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch })); expect(captured().body.serviceTier).toBeUndefined(); }); }); diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index f5ef242be..1ca718188 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -26,8 +26,6 @@ async function captureGooglePayload( await streamGoogle(model, context, { apiKey: "test-key", - // Capture the generateContent request shape; gemini-3 ids auto-route to Interactions by default. - useInteractionsApi: false, onPayload: payload => { captured = payload as { config: { systemInstruction?: unknown }; contents: unknown[] }; }, diff --git a/packages/ai/test/issue-1270-repro.test.ts b/packages/ai/test/issue-1270-repro.test.ts index 2154ff79a..5d3582254 100644 --- a/packages/ai/test/issue-1270-repro.test.ts +++ b/packages/ai/test/issue-1270-repro.test.ts @@ -44,8 +44,6 @@ describe("issue #1270: Vertex AI global endpoint", () => { const stream = streamGoogleVertex(model, context, { project: "vertex-project", location: "global", - // This asserts the generateContent URL; gemini-3 ids auto-route to Interactions by default. - useInteractionsApi: false, fetch: async input => { const url = input instanceof Request ? input.url : input.toString(); urls.push(url); diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 08d3eee01..cd7837a15 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -109,7 +109,7 @@ import { obfuscateProviderContext, SecretObfuscator, } from "./secrets"; -import { AgentSession, type Prewalk, type PlanYolo } from "./session/agent-session"; +import { AgentSession, type PlanYolo, type Prewalk } from "./session/agent-session"; import { discoverAuthStorage as discoverAuthStorageFromConfig } from "./session/auth-broker-config"; import type { AuthStorage } from "./session/auth-storage"; import { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 68ac6c306..9065f723f 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -259,9 +259,6 @@ import goalModeContextPrompt from "../prompts/goals/goal-mode-context.md" with { import goalTodoContextPrompt from "../prompts/goals/goal-todo-context.md" with { type: "text" }; import parentIrcSteerTemplate from "../prompts/steering/parent-irc.md" with { type: "text" }; import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type: "text" }; -import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; -import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; -import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; @@ -276,6 +273,9 @@ import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool type: "text", }; import planYoloHandoffPrompt from "../prompts/system/plan-yolo-handoff.md" with { type: "text" }; +import prewalkChecklistPrompt from "../prompts/system/prewalk-checklist.md" with { type: "text" }; +import prewalkContinuePrompt from "../prompts/system/prewalk-continue.md" with { type: "text" }; +import prewalkPlanPrompt from "../prompts/system/prewalk-plan.md" with { type: "text" }; import rewindReportTemplate from "../prompts/system/rewind-report.md" with { type: "text" }; import sideChannelNoToolsReminder from "../prompts/system/side-channel-no-tools.md" with { type: "text" }; import thinkingLoopRedirectTemplate from "../prompts/system/thinking-loop-redirect.md" with { type: "text" }; diff --git a/packages/coding-agent/test/agent-session-prune-persistence.test.ts b/packages/coding-agent/test/agent-session-prune-persistence.test.ts index 3b78c3657..438f0de4e 100644 --- a/packages/coding-agent/test/agent-session-prune-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-prune-persistence.test.ts @@ -114,7 +114,7 @@ describe("AgentSession per-turn prune persistence", () => { const message = session.agent.state.messages.find( candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID, ); - if (!message || message.role !== "toolResult" || !Array.isArray(message.content)) { + if (message?.role !== "toolResult" || !Array.isArray(message.content)) { throw new Error("Expected the seeded tool result in live agent state"); } const text = message.content.find(block => block.type === "text"); @@ -156,7 +156,7 @@ describe("AgentSession per-turn prune persistence", () => { const rebuilt = reloaded .buildSessionContext() .messages.find(candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID); - if (!rebuilt || rebuilt.role !== "toolResult" || !Array.isArray(rebuilt.content)) { + if (rebuilt?.role !== "toolResult" || !Array.isArray(rebuilt.content)) { throw new Error("Expected the seeded tool result in the from-disk rebuild"); } const rebuiltText = rebuilt.content.find(block => block.type === "text"); From 3e5b7da6f05a7ab14e748ad42f3dd80c261d985d Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:48:20 +0200 Subject: [PATCH 14/28] feat(auth-gateway): added diagnostic headers to auth-gateway inference responses - Added a `gatewayResponseHeaders` helper and an optional `json()` headers parameter to attach gateway diagnostics to HTTP responses. - Updated auth-gateway format and Pi-native handlers to generate request IDs and include them with model metadata in all responses, while adding cost and timing headers when non-streaming responses include a final assistant message. - Added tests validating header presence for streaming vs non-streaming responses and documented the new auth-gateway diagnostic headers in the unreleased changelog. --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-gateway/http.ts | 37 ++++++- packages/ai/src/auth-gateway/server.ts | 26 ++++- .../auth-gateway-response-headers.test.ts | 103 ++++++++++++++++++ 4 files changed, 165 insertions(+), 5 deletions(-) create mode 100644 packages/ai/test/auth-gateway-response-headers.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 03fa9035e..712cd2e4f 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added diagnostic response headers to auth-gateway inference endpoints: `x-request-id`/`request-id` (correlates with gateway logs; surfaced by OpenAI/Anthropic SDKs) and LiteLLM-style `x-litellm-model-id`/`x-litellm-model-api-base` on every response, plus `x-litellm-response-cost`, `x-litellm-response-duration-ms`, and `openai-processing-ms` on non-streaming responses + ### Changed - Switched Google and Google Vertex providers to always use `streamGenerateContent` requests diff --git a/packages/ai/src/auth-gateway/http.ts b/packages/ai/src/auth-gateway/http.ts index 21ea80d61..ec905a308 100644 --- a/packages/ai/src/auth-gateway/http.ts +++ b/packages/ai/src/auth-gateway/http.ts @@ -5,19 +5,50 @@ * and peer-resolution logic. */ import { timingSafeEqual as nodeTimingSafeEqual } from "node:crypto"; +import type { Api, AssistantMessage, Model } from "../types"; const JSON_HEADERS = { "Content-Type": "application/json", "X-Content-Type-Options": "nosniff", } as const; -export function json(status: number, body: unknown): Response { +export function json(status: number, body: unknown, headers?: Record): Response { return new Response(JSON.stringify(body) ?? "null", { status, - headers: JSON_HEADERS, + headers: headers ? { ...JSON_HEADERS, ...headers } : JSON_HEADERS, }); } +/** + * Diagnostic response headers for translated inference requests, mirroring the + * names existing gateway-aware clients already parse: `x-request-id` / + * `request-id` (surfaced as `_request_id` by the OpenAI and Anthropic SDKs, + * matches the gateway log line), LiteLLM's model-resolution and cost headers, + * and OpenAI's `openai-processing-ms`. Model/request-id headers are always + * present; `message` — the final assistant message, available only on + * non-streaming responses — adds the computed cost, and `startedAt` the wall + * time. Streaming responses send headers before usage exists, so they carry + * only the identity headers. + */ +export function gatewayResponseHeaders( + model: Model, + info: { requestId: string; message?: AssistantMessage; startedAt?: number }, +): Record { + const headers: Record = { + "x-request-id": info.requestId, + "request-id": info.requestId, + "x-litellm-model-id": model.id, + }; + if (model.baseUrl) headers["x-litellm-model-api-base"] = model.baseUrl; + if (info.message) headers["x-litellm-response-cost"] = info.message.usage.cost.total.toString(); + if (info.startedAt !== undefined) { + const elapsed = (performance.now() - info.startedAt).toFixed(0); + headers["x-litellm-response-duration-ms"] = elapsed; + headers["openai-processing-ms"] = elapsed; + } + return headers; +} + export function resolvePeer(req: Request): string { const fwd = req.headers.get("x-forwarded-for"); if (fwd) return fwd.split(",")[0].trim(); @@ -165,6 +196,8 @@ const CORS_HEADERS: Record = { "Access-Control-Allow-Methods": "GET, POST, OPTIONS", "Access-Control-Allow-Headers": "authorization, content-type, anthropic-version, anthropic-beta, openai-organization, openai-project, x-stainless-*, x-api-key", + "Access-Control-Expose-Headers": + "x-request-id, request-id, x-litellm-model-id, x-litellm-model-api-base, x-litellm-response-cost, x-litellm-response-duration-ms, openai-processing-ms", "Access-Control-Max-Age": "86400", }; diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index ae4d47415..2624dd028 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -32,7 +32,15 @@ import { completeSimple, streamSimple } from "../stream"; import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types"; import { deterministicUuid } from "../utils/deterministic-id"; import { parseBind } from "../utils/parse-bind"; -import { captureRequestHeaders, corsHeaders, isAuthorized, json, resolvePeer, withCors } from "./http"; +import { + captureRequestHeaders, + corsHeaders, + gatewayResponseHeaders, + isAuthorized, + json, + resolvePeer, + withCors, +} from "./http"; import type { AuthGatewayServerHandle, AuthGatewayServerOptions, @@ -334,6 +342,8 @@ async function handleFormatEndpoint( req: Request, peer: string, ): Promise { + const startedAt = performance.now(); + const requestId = crypto.randomUUID(); const controller = mirrorRequestAbort(req); if (controller.signal.aborted) return clientClosedResponse(route); @@ -430,6 +440,7 @@ async function handleFormatEndpoint( ); logger.info("auth-gateway request", { + requestId, format: route.label, model: parsed.modelId, resolvedProvider: model.provider, @@ -458,7 +469,11 @@ async function handleFormatEndpoint( const classified = classifyGatewayError(errorMessage); return route.module.formatError(classified.status, classified.type, errorMessage); } - return json(200, route.module.encodeResponse(message, parsed.modelId)); + return json( + 200, + route.module.encodeResponse(message, parsed.modelId), + gatewayResponseHeaders(model, { requestId, message, startedAt }), + ); } catch (error) { if (controller.signal.aborted) return clientClosedResponse(route); const classified = classifyGatewayError(error); @@ -493,6 +508,7 @@ async function handleFormatEndpoint( return new Response(sseStream, { status: 200, headers: { + ...gatewayResponseHeaders(model, { requestId }), "Content-Type": "text/event-stream; charset=utf-8", "Cache-Control": "no-cache", Connection: "keep-alive", @@ -519,6 +535,8 @@ async function handleFormatEndpoint( * path. */ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise { + const startedAt = performance.now(); + const requestId = crypto.randomUUID(); const controller = mirrorRequestAbort(req); const aborted = (): Response => piNative.formatError(499, "request_aborted", "client closed request"); if (controller.signal.aborted) return aborted(); @@ -605,6 +623,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe streamOpts.sessionId ??= sessionId; logger.info("auth-gateway request", { + requestId, format: "pi-native", model: parsed.modelId, resolvedProvider: model.provider, @@ -633,7 +652,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe const classified = classifyGatewayError(errorMessage); return piNative.formatError(classified.status, classified.type, errorMessage); } - return json(200, { message }); + return json(200, { message }, gatewayResponseHeaders(model, { requestId, message, startedAt })); } catch (error) { if (controller.signal.aborted) return aborted(); const classified = classifyGatewayError(error); @@ -664,6 +683,7 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe return new Response(sseStream, { status: 200, headers: { + ...gatewayResponseHeaders(model, { requestId }), "Content-Type": "text/event-stream; charset=utf-8", "Cache-Control": "no-cache", Connection: "keep-alive", diff --git a/packages/ai/test/auth-gateway-response-headers.test.ts b/packages/ai/test/auth-gateway-response-headers.test.ts new file mode 100644 index 000000000..986d5c084 --- /dev/null +++ b/packages/ai/test/auth-gateway-response-headers.test.ts @@ -0,0 +1,103 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry"; +import { startAuthGateway } from "@oh-my-pi/pi-ai/auth-gateway"; +import { AuthStorage } from "@oh-my-pi/pi-ai/auth-storage"; +import { createMockModel, type MockModel, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock"; + +interface GatewayHarness { + url: string; + mock: MockModel; + close(): Promise; +} + +async function bootGateway(): Promise { + registerMockApi(); + const dir = await fs.mkdtemp(path.join(os.tmpdir(), "gw-response-headers-")); + const storage = await AuthStorage.create(path.join(dir, "auth.db")); + storage.setRuntimeApiKey("openrouter", "test-key"); + const mock = createMockModel({ provider: "openrouter", id: "mock/header-model" }); + const handle = startAuthGateway({ + bind: "127.0.0.1:0", + bearerTokens: ["t"], + storage, + resolveModel: () => mock.model, + version: "test", + }); + return { + url: handle.url, + mock, + close: async () => { + await handle.close(); + storage.close(); + await fs.rm(dir, { recursive: true, force: true }); + }, + }; +} + +afterEach(() => { + clearCustomApis(); +}); + +describe("auth-gateway diagnostic response headers", () => { + it("non-streaming responses carry cost, model id, request id, and duration", async () => { + const gw = await bootGateway(); + try { + gw.mock.push({ + content: ["hello"], + usage: { + input: 10, + output: 5, + totalTokens: 15, + cost: { input: 0.001, output: 0.0002, total: 0.0012 }, + }, + }); + const res = await fetch(`${gw.url}/v1/chat/completions`, { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: "Bearer t" }, + body: JSON.stringify({ + model: "mock/header-model", + messages: [{ role: "user", content: "hi" }], + stream: false, + }), + }); + expect(res.status).toBe(200); + expect(res.headers.get("x-litellm-response-cost")).toBe("0.0012"); + expect(res.headers.get("x-litellm-model-id")).toBe("mock/header-model"); + const duration = res.headers.get("x-litellm-response-duration-ms"); + expect(duration).not.toBeNull(); + expect(Number(duration)).toBeGreaterThanOrEqual(0); + expect(res.headers.get("openai-processing-ms")).toBe(duration); + const requestId = res.headers.get("x-request-id"); + expect(requestId).toMatch(/^[0-9a-f-]{36}$/); + expect(res.headers.get("request-id")).toBe(requestId); + } finally { + await gw.close(); + } + }); + + it("streaming responses carry the model and request ids but no cost (unknown at header time)", async () => { + const gw = await bootGateway(); + try { + gw.mock.push({ content: ["hello"] }); + const res = await fetch(`${gw.url}/v1/chat/completions`, { + method: "POST", + headers: { "Content-Type": "application/json", Authorization: "Bearer t" }, + body: JSON.stringify({ + model: "mock/header-model", + messages: [{ role: "user", content: "hi" }], + stream: true, + }), + }); + expect(res.status).toBe(200); + expect(res.headers.get("x-litellm-model-id")).toBe("mock/header-model"); + expect(res.headers.get("x-request-id")).toMatch(/^[0-9a-f-]{36}$/); + expect(res.headers.get("x-litellm-response-cost")).toBeNull(); + await res.text(); + } finally { + await gw.close(); + } + }); +}); From 883e68f2d2694554b13e429d7ffeee054497b312 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:50:24 +0200 Subject: [PATCH 15/28] deps(deps): bumped dependency versions and refreshed patch mappings - Upgraded `@agentclientprotocol/sdk` from `0.25.0` to `1.2.1` and `@ark/schema` from `0.56.1` to `0.56.2` in `package.json` and `bun.lock`, updating `patchedDependencies` and override entries accordingly. - Updated lockfile package metadata to match new manifest versions, including TypeScript `7.0.2`, arktype `2.2.3`, and several related dependency/version dependency-tree entries. - Added a new `@agentclientprotocol/sdk@1.2.1` patch that extends the package exports with `./dist/schema/zod.gen.js` including type and default import mappings. --- bun.lock | 159 ++++++++++-------- package.json | 55 +++--- packages/coding-agent/CHANGELOG.md | 1 + .../@agentclientprotocol%2Fsdk@1.2.1.patch | 18 ++ ....56.1.patch => @ark%2Fschema@0.56.2.patch} | 0 5 files changed, 136 insertions(+), 97 deletions(-) create mode 100644 patches/@agentclientprotocol%2Fsdk@1.2.1.patch rename patches/{@ark%2Fschema@0.56.1.patch => @ark%2Fschema@0.56.2.patch} (100%) diff --git a/bun.lock b/bun.lock index e57ea34e3..610d212c0 100644 --- a/bun.lock +++ b/bun.lock @@ -341,24 +341,25 @@ }, }, "patchedDependencies": { - "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", + "@ark/schema@0.56.2": "patches/@ark%2Fschema@0.56.2.patch", + "@agentclientprotocol/sdk@1.2.1": "patches/@agentclientprotocol%2Fsdk@1.2.1.patch", }, "overrides": { - "@ark/schema": "0.56.1", + "@ark/schema": "0.56.2", }, "catalog": { - "@agentclientprotocol/sdk": "0.25.0", + "@agentclientprotocol/sdk": "1.2.1", "@babel/generator": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.5", - "@bufbuild/protobuf": "^2.12.0", - "@bufbuild/protoc-gen-es": "^2.12.0", + "@biomejs/biome": "^2.5.3", + "@bufbuild/protobuf": "^2.12.1", + "@bufbuild/protoc-gen-es": "^2.12.1", "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", - "@napi-rs/cli": "3.7.0", + "@napi-rs/cli": "3.7.2", "@oh-my-pi/hashline": "16.4.8", "@oh-my-pi/omp-stats": "16.4.8", "@oh-my-pi/pi-agent-core": "16.4.8", @@ -373,61 +374,61 @@ "@oh-my-pi/snapcompact": "16.4.8", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", - "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", + "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", "@opentelemetry/resources": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", "@opentelemetry/sdk-trace-node": "^2.7.1", - "@puppeteer/browsers": "^3.0.4", - "@tailwindcss/node": "^4.3.0", - "@tailwindcss/vite": "^4.3.0", + "@puppeteer/browsers": "^3.0.6", + "@tailwindcss/node": "^4.3.2", + "@tailwindcss/vite": "^4.3.2", "@types/babel__generator": "^7.27.0", "@types/babel__traverse": "^7.28.0", "@types/bun": "^1.3.14", "@types/react": "^19.2.17", "@types/react-dom": "^19.2.3", "@types/turndown": "5.0.6", - "@typescript/native-preview": "7.0.0-dev.20260609.1", + "@typescript/native-preview": "7.0.0-dev.20260707.2", "@xterm/headless": "^6.0.0", - "arktype": "2.2.2", + "arktype": "2.2.3", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", "diff": "^9.0.0", - "fast-xml-parser": "^5.9.0", + "fast-xml-parser": "^5.9.3", "fastembed": "2.1.0", "fflate": "0.8.3", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", "header-generator": "^2.1.82", - "linkedom": "^0.18.12", - "lint-staged": "^17.0.7", - "lru-cache": "11.5.1", - "lucide-react": "^1.17.0", + "linkedom": "^0.18.13", + "lint-staged": "^17.0.8", + "lru-cache": "11.5.2", + "lucide-react": "^1.24.0", "mammoth": "^1.12.0", - "marked": "^18.0.5", - "mupdf": "^1.27.0", + "marked": "^18.0.6", + "mupdf": "^1.28.0", "onnxruntime-node": "1.26.0", - "postcss": "^8.5.15", - "prettier": "^3.8.4", + "postcss": "^8.5.16", + "prettier": "^3.9.5", "puppeteer-core": "25.3.0", "react": "19.2.7", "react-chartjs-2": "^5.3.1", "react-dom": "19.2.7", "regexp-tree": "^0.1.27", - "solid-js": "^1.9.13", - "tailwindcss": "^4.3.0", + "solid-js": "^1.9.14", + "tailwindcss": "^4.3.2", "ts-morph": "^28.0.0", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", - "typescript": "^6.0.3", - "vite": "^8.0.16", + "typescript": "^7.0.2", + "vite": "^8.1.4", "vite-plugin-solid": "^2.11.12", "winston": "^3.19.0", "winston-daily-rotate-file": "^5.0.0", "zod": "^4", }, "packages": { - "@agentclientprotocol/sdk": ["@agentclientprotocol/sdk@0.25.0", "", { "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" } }, "sha512-wU1VgXNtMvdVotX49txc3WJUDV+/QbLpsgjMvFhlRmp37osdLbI7L7y+iwAlQATwfjLxcv1r1p3ZxZBcXlGhcQ=="], + "@agentclientprotocol/sdk": ["@agentclientprotocol/sdk@1.2.1", "", { "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" } }, "sha512-jwYUdOQR7tc+Zfch53VL4JJyUNK/46q03uUTYb+PjECsmnNl94XFXOfYLJ8RBpMNidXd1rpOAVgb0vqD98xImA=="], "@anush008/tokenizers": ["@anush008/tokenizers@0.0.0", "", { "optionalDependencies": { "@anush008/tokenizers-darwin-universal": "0.0.0", "@anush008/tokenizers-linux-x64-gnu": "0.0.0", "@anush008/tokenizers-win32-x64-msvc": "0.0.0" } }, "sha512-IQD9wkVReKAhsEAbDjh/0KrBGTEXelqZLpOBRDaIRvlzZ9sjmUP+gKbpvzyJnei2JHQiE8JAgj7YcNloINbGBw=="], @@ -437,9 +438,9 @@ "@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="], - "@ark/schema": ["@ark/schema@0.56.1", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-1Cf2g9nKD8K/3JGRu+gCCfYw5d4qR8YLLjDs5W5kpmaButCYWAPFUJqSXyBATPjglzCd4tIkp398iPYVs8MjRA=="], + "@ark/schema": ["@ark/schema@0.56.2", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-Qx4D2JFbBWpntiHZaTv7bGG4H/M2rigiknezKg/WVyDSaLdE4YCcWAOoFB7pjjDqHbbV2OqRfntm1nnXvwMexg=="], - "@ark/util": ["@ark/util@0.56.1", "", {}, "sha512-Tp1rTik3q5Z+jAeeDxr5JZpmVIw0miti1ykSEHyZv5Pw3TIJl2xbN6KTacOxITp0l3s9ytlrWd30Zvqcy5vzoQ=="], + "@ark/util": ["@ark/util@0.56.2", "", {}, "sha512-9kU2sUE38FZEGG7l3hamYMBieLYEJh2L1mrYD2eXpT+78EnQSV1bhjxJhnxGBMSTbtwpBSDNSK+K60WvaI/DTQ=="], "@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="], @@ -625,7 +626,7 @@ "@mozilla/readability": ["@mozilla/readability@0.6.0", "", {}, "sha512-juG5VWh4qAivzTAeMzvY9xs9HY5rAcr2E4I7tiSSCokRFi7XIZCAu92ZkSTsIj1OPceCifL3cpfteP3pDT9/QQ=="], - "@napi-rs/cli": ["@napi-rs/cli@3.7.0", "", { "dependencies": { "@inquirer/prompts": "^8.0.0", "@napi-rs/cross-toolchain": "^1.0.3", "@napi-rs/wasm-tools": "^1.0.1", "@octokit/rest": "^22.0.1", "clipanion": "^4.0.0-rc.4", "colorette": "^2.0.20", "emnapi": "^1.10.0", "es-toolkit": "^1.41.0", "js-yaml": "^4.1.0", "obug": "^2.0.0", "semver": "^7.7.3", "typanion": "^3.14.0" }, "peerDependencies": { "@emnapi/runtime": "^1.7.1" }, "optionalPeers": ["@emnapi/runtime"], "bin": { "napi": "dist/cli.js", "napi-raw": "cli.mjs" } }, "sha512-3d3+rmxlOIV/G1zPWeX4PCxuYnhcCQM2BvY9rtimC8RO0dFR9gtYP+Grov+WoduZtfWRj5N1XvytWeRxxCk5zw=="], + "@napi-rs/cli": ["@napi-rs/cli@3.7.2", "", { "dependencies": { "@inquirer/prompts": "^8.5.2", "@napi-rs/cross-toolchain": "^1.0.3", "@napi-rs/wasm-tools": "^1.0.1", "@octokit/rest": "^22.0.1", "clipanion": "^4.0.0-rc.4", "colorette": "^2.0.20", "emnapi": "^1.11.1", "es-toolkit": "^1.47.0", "js-yaml": "^4.2.0", "obug": "^2.1.2", "semver": "^7.8.2", "typanion": "^3.14.0" }, "peerDependencies": { "@emnapi/runtime": "^1.7.1" }, "optionalPeers": ["@emnapi/runtime"], "bin": { "napi": "dist/cli.js", "napi-raw": "cli.mjs" } }, "sha512-shDW0Td/XZQpP04Yy+OsMt1ILMKGGkoLcy1zVAsSAK0fLfWm0Upgkmfs/NOV2ZhMQwkgpR3ZEdyHmTwgrUDQuA=="], "@napi-rs/cross-toolchain": ["@napi-rs/cross-toolchain@1.0.3", "", { "dependencies": { "@napi-rs/lzma": "^1.4.5", "@napi-rs/tar": "^1.1.0", "debug": "^4.4.1" }, "peerDependencies": { "@napi-rs/cross-toolchain-arm64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-arm64-target-x86_64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-aarch64": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-armv7": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-ppc64le": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-s390x": "^1.0.3", "@napi-rs/cross-toolchain-x64-target-x86_64": "^1.0.3" }, "optionalPeers": ["@napi-rs/cross-toolchain-arm64-target-aarch64", "@napi-rs/cross-toolchain-arm64-target-armv7", "@napi-rs/cross-toolchain-arm64-target-ppc64le", "@napi-rs/cross-toolchain-arm64-target-s390x", "@napi-rs/cross-toolchain-arm64-target-x86_64", "@napi-rs/cross-toolchain-x64-target-aarch64", "@napi-rs/cross-toolchain-x64-target-armv7", "@napi-rs/cross-toolchain-x64-target-ppc64le", "@napi-rs/cross-toolchain-x64-target-s390x", "@napi-rs/cross-toolchain-x64-target-x86_64"] }, "sha512-ENPfLe4937bsKVTDA6zdABx4pq9w0tHqRrJHyaGxgaPq03a2Bd1unD5XSKjXJjebsABJ+MjAv1A2OvCgK9yehg=="], @@ -789,23 +790,23 @@ "@opentelemetry/api": ["@opentelemetry/api@1.9.1", "", {}, "sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q=="], - "@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-fmEWp5kXlGEc3i/lR698Hz41DfGyN4Tbe4g7L1AxSc7fF8Xeh/FQ9Quqpa9dVA413Q1Ad43QOLzU4JoXgbFPWw=="], + "@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.220.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CmVa4ImJ+ynfrPMNaAXHET6Bhb44SwzmfyVJFq9ni2jgXJR/l7C6gfVFddNmHP+ZOkP9cf4f9DBe68qVLTHc9w=="], "@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.9.0", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-OQ0vzvbZBiUhjqLnUaoNfYmP8553Crr3aggB4y0ZUi815mZ7idpdJXQmoKdeBKJelYttoBlLSSHubmyw3wvX4w=="], "@opentelemetry/core": ["@opentelemetry/core@2.9.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw=="], - "@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.218.0", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/otlp-exporter-base": "0.218.0", "@opentelemetry/otlp-transformer": "0.218.0", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-r1Msf8SNLRmwh9J6XQ5uh82D7CdDWMNHnPB7LAVHjzut0TkSeKc5KcIvr4SvHvfk/xwN5gxC+VLKQ1k0o8PSPw=="], + "@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-exporter-base": "0.220.0", "@opentelemetry/otlp-transformer": "0.220.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-trace": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-voTAD8XgJxlK7zLkXh8EzMB09zrQr3tyY/BsnDTlDiQU/UdK58MZ63A3mUjdEDrxMjCVmBHU3WQJhRmQe+Dvzg=="], - "@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.218.0", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/otlp-transformer": "0.218.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-ZwqpkNL5W7RyGJPDZ9g06DvKp8KFTWPJPN12anpMQYSKpTSU0z3EIZuPq9vPGpS8siFyOqDYDAuCwlNO9FqgbA=="], + "@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.220.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/otlp-transformer": "0.220.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CXYo8UD5Mn9YbgebO2EL4wejtA+gxLmLiu6HCk2KH2BR7XhFN6/6p1UlCb23DYCjeYkndevLHuejCCN1yx4+OQ=="], - "@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-logs": "0.218.0", "@opentelemetry/sdk-metrics": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CFaKH87WAzjuJ4awowTTLzUvMfaRfiOFG5+qm5S5ncyalRtN4ecQ+YmuANJSCrVPuvZFEkUgKhBPBndxi3rHsQ=="], + "@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.220.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.220.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/sdk-logs": "0.220.0", "@opentelemetry/sdk-metrics": "2.9.0", "@opentelemetry/sdk-trace": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-lXGrv7KXZ0gNH9SVNUaa6vv6phVYGvJxfXAlMbzbakiXru75f5MZl8Z7oqiMMQD77riVHJCFlQvbZs/VVN2/4A=="], "@opentelemetry/resources": ["@opentelemetry/resources@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-jyA5MBLQ+Dkl3+JsZkUoUvL7yHvU64kLsvpXKarWm6347Sl1t1bXFTFykUePNpT5WH5pm9a2Qtt03iIYQhZ1Fg=="], - "@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-QvnNdugatFTVCJXH0Mcu7GOOJSylA9j127kIezOE4YwTI4YbowRons2K4WZTv5FMS8T4q9P0NdaRHdkSmeAIag=="], + "@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.220.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.220.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-WywcTkQtv2iNmt+6y5Kcd4rzvx9bLVsBa2Nwcmg01IUaBTkTow3W4d9KE5vNBpEDtb9tp21WcRBY/lANRrApYA=="], - "@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-MpDJdkiFDs3Pm1RHO3KByuZbuBdJEXEAkiC0+yJdsZGVCdf1RpHR6n+LHDcS7ffmfrt5kVCzJSCfm4z2C7v0uQ=="], + "@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-Xx8RGS4H5XEBl01WuCreMIpiah9cCXMbSkeuIePPdD2cUpq/vUzYmj8E/MK1OsbOc93FuAD4jfn2WOacKwLn7Q=="], "@opentelemetry/sdk-trace": ["@opentelemetry/sdk-trace@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/resources": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-sGA19HvtrrSKYsseHphluH6j3p6Xa3fqc7c7y8f/7mYWejc1lyDFcpSdD1kYa50HCLUeEo4zA5bW0pniaPszuw=="], @@ -935,21 +936,61 @@ "@types/turndown": ["@types/turndown@5.0.6", "", {}, "sha512-ru00MoyeeouE5BX4gRL+6m/BsDfbRayOskWqUvh7CLGW+UXxHQItqALa38kKnOiZPqJrtzJUgAC2+F0rL1S4Pg=="], - "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260609.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260609.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260609.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260609.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-1HOuH/u/451O3hx4Z9fesNqarpeit6UfkgwK96sCVWi5p69F0N3v+6bI969lLIjF7K9dbYQNiWUaZ6Wik87iKg=="], + "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260707.2", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260707.2", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260707.2", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260707.2" }, "bin": { "tsgo": "bin/tsgo" } }, "sha512-oUGp+Rep/hqMhPunyinsALUwSlzHINSxitifPiSaeqoKOKD2OlR9NE3TaPqwsl4NlGslsOSUXI1JotWQzpYCPg=="], - "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260609.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-Yf/zHEadP/yUiWUdM/mZVfEVFJuGMf6nhRSFif0vp+FwtfGU4jmlpNF7BTJJdOHrrcWkwEJKzAoMCtEtyxhuyQ=="], + "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260707.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-wny2pgKjGbiZtnOIHVa3tXC1UfDqxNEFzyPGmiqybedG8hipG2Nfp0l5UxbaKCjkLacUpH/W5bP2hBOMVhCOzg=="], - "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260609.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-z4dYWI57CPHs0wV/FWFth8fWmqYH7iOm7THOfZ5Fv0jo/SWK6kE1kEUIqIAExqo7ueRNqSrCw0I8U1J4TJszAw=="], + "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260707.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-Afc7M5zOwo+GpfcYwz5Z8HMB2tPVsui7nNIqEuuFB73MPdVqNn/Wmpe4tP4MRri0AtJnJknoHBaTJ/VDAp/Jhw=="], - "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "arm" }, "sha512-mEtN8BbAgVtBu/5MVomYquXNvgok2C0KG6V0D4SV1jfBJNtlcqbp0WuIqT0bnM9DA4TgzcHvnFMpwGSK/dqI5A=="], + "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "arm" }, "sha512-hJm/UOqZTr9FHmR7uNm8VGX4oKtfWk0Jem0zPeJFNC8ckGUfSBueyiEYMZB+XmRc1aG4x1E46y3CplP4CLHvGQ=="], - "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-OxNVWH9IhrMAzNlDyDit1dPO64GFIDPOUKoruIkJ9A1ZEONfIHXG5f+V3si9jtuNmuomiz9FjpbzOqLsgaxt+w=="], + "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-iITBa2WjjTI5N9t5l7Z4KoOSI+2zBlhbvFzsD/f8qX8QoKjz/Y4DPyBDgezYi8nkqjjksbgSOJ3/ykzhwrB9cg=="], - "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260609.1", "", { "os": "linux", "cpu": "x64" }, "sha512-KO8WO1gBIC09T3255RlTY42TGu8en5mEkLPQu2wkMn+dX2T8KYL64zXrCeLeUWa0NvmVdJUeyWu3pFOn3zKemw=="], + "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260707.2", "", { "os": "linux", "cpu": "x64" }, "sha512-du0dzi6y97Po5vDNdPJTyyijHCpaS22JLRnKZEJXBDaO9gCIymOv/5QQokFRuOlQm0bWl3i9PF4OVdGP6uAOQA=="], - "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260609.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-+8q19LWjnMKK6SF3PLeMEalbfWDYWHs0AU8kSFCBCke/RLoDG4FjQzVtLgUo+KWhsmZMosiEyqEnZmSlED2tIQ=="], + "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260707.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-SsAwfhyHJ1akgBc+99z4+hwdbHsdWaKB8EwCNIMA6JfSLMeUjffrYvxu+vfMyxVtOVOz7RrRXRoiDiu4a2sCtg=="], - "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260609.1", "", { "os": "win32", "cpu": "x64" }, "sha512-qNPcss+6yRoNFfFIKQbPwJWYxDfOZwyL8JBJh4J+yMLOad/+/AOjsO4EtZsIpv5PMCjpnD75coBoDkw+5NkItw=="], + "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260707.2", "", { "os": "win32", "cpu": "x64" }, "sha512-DL4u27stv0fo71sVhOzHSwE+YMZsbBijVI+kg5dLDLilSH79WFTJ8RSQ46vJrCMt+Gjlv/JOZP1PuLJDfioYeQ=="], + + "@typescript/typescript-aix-ppc64": ["@typescript/typescript-aix-ppc64@7.0.2", "", { "os": "aix", "cpu": "ppc64" }, "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ=="], + + "@typescript/typescript-darwin-arm64": ["@typescript/typescript-darwin-arm64@7.0.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA=="], + + "@typescript/typescript-darwin-x64": ["@typescript/typescript-darwin-x64@7.0.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA=="], + + "@typescript/typescript-freebsd-arm64": ["@typescript/typescript-freebsd-arm64@7.0.2", "", { "os": "freebsd", "cpu": "arm64" }, "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ=="], + + "@typescript/typescript-freebsd-x64": ["@typescript/typescript-freebsd-x64@7.0.2", "", { "os": "freebsd", "cpu": "x64" }, "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw=="], + + "@typescript/typescript-linux-arm": ["@typescript/typescript-linux-arm@7.0.2", "", { "os": "linux", "cpu": "arm" }, "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ=="], + + "@typescript/typescript-linux-arm64": ["@typescript/typescript-linux-arm64@7.0.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ=="], + + "@typescript/typescript-linux-loong64": ["@typescript/typescript-linux-loong64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ=="], + + "@typescript/typescript-linux-mips64el": ["@typescript/typescript-linux-mips64el@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA=="], + + "@typescript/typescript-linux-ppc64": ["@typescript/typescript-linux-ppc64@7.0.2", "", { "os": "linux", "cpu": "ppc64" }, "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA=="], + + "@typescript/typescript-linux-riscv64": ["@typescript/typescript-linux-riscv64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ=="], + + "@typescript/typescript-linux-s390x": ["@typescript/typescript-linux-s390x@7.0.2", "", { "os": "linux", "cpu": "s390x" }, "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw=="], + + "@typescript/typescript-linux-x64": ["@typescript/typescript-linux-x64@7.0.2", "", { "os": "linux", "cpu": "x64" }, "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A=="], + + "@typescript/typescript-netbsd-arm64": ["@typescript/typescript-netbsd-arm64@7.0.2", "", { "os": "none", "cpu": "arm64" }, "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA=="], + + "@typescript/typescript-netbsd-x64": ["@typescript/typescript-netbsd-x64@7.0.2", "", { "os": "none", "cpu": "x64" }, "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA=="], + + "@typescript/typescript-openbsd-arm64": ["@typescript/typescript-openbsd-arm64@7.0.2", "", { "os": "openbsd", "cpu": "arm64" }, "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ=="], + + "@typescript/typescript-openbsd-x64": ["@typescript/typescript-openbsd-x64@7.0.2", "", { "os": "openbsd", "cpu": "x64" }, "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg=="], + + "@typescript/typescript-sunos-x64": ["@typescript/typescript-sunos-x64@7.0.2", "", { "os": "sunos", "cpu": "x64" }, "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g=="], + + "@typescript/typescript-win32-arm64": ["@typescript/typescript-win32-arm64@7.0.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ=="], + + "@typescript/typescript-win32-x64": ["@typescript/typescript-win32-x64@7.0.2", "", { "os": "win32", "cpu": "x64" }, "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g=="], "@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, "sha512-PJFXFS4ZJKiJ9Qiuix6Dz/OwEIqHD7Dme1UwZhTK11vR+5dqW2ACbdndWQexBzCx+CPuMe5WBYQWCsFyGlQLlQ=="], @@ -969,9 +1010,9 @@ "argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="], - "arkregex": ["arkregex@0.0.7", "", { "dependencies": { "@ark/util": "0.56.1" } }, "sha512-O/Ltrn9EUSn3ui0KVzfyrWGDUsHlzKxDVBtpQxL/6JmLRMAZAebfSNf/A/J5Ny5S6QIwrXX+RfXsu888HMs35A=="], + "arkregex": ["arkregex@0.0.8", "", { "dependencies": { "@ark/util": "0.56.2" } }, "sha512-PJcx6G1kQTgLKPUbeYlYecDRaKq15AMSGVajlKFYWlPeJRQL+j3dKE6tyMs40HZ99djS1l9Vhl3ezAHy9JBIqQ=="], - "arktype": ["arktype@2.2.2", "", { "dependencies": { "@ark/schema": "0.56.1", "@ark/util": "0.56.1", "arkregex": "0.0.7" } }, "sha512-YYf1xhL2dh5aPZFlsY0RAsxv5HZqfLGLptH2ZP3JidTmsGRW8VOymhPjjMTkerL12vR2YtX0SK4c1mATtae8SA=="], + "arktype": ["arktype@2.2.3", "", { "dependencies": { "@ark/schema": "0.56.2", "@ark/util": "0.56.2", "arkregex": "0.0.8" } }, "sha512-7W+0RLTUNJiBFIIZXwOQxSR8Z273IAd6IvqBeG9+gHnQKFsIx2C0iOtGTmMrPnlX4qLXyc5+ll7A0BIj9WrbTg=="], "async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="], @@ -1271,7 +1312,7 @@ "lop": ["lop@0.4.2", "", { "dependencies": { "duck": "^0.1.12", "option": "~0.2.1", "underscore": "^1.13.1" } }, "sha512-RefILVDQ4DKoRZsJ4Pj22TxE3omDO47yFpkIBoDKzkqPRISs5U1cnAdg/5583YPkWPaLIYHOKRMQSvjFsO26cw=="], - "lru-cache": ["lru-cache@11.5.1", "", {}, "sha512-RPimw/7aMdv2oqRrxKwvZXcPfwBrn/JZ2xYcY9Hus/6LaS3VOAKVWKWgNLCFSiOm1ESXinjsDlidVU7JlnCN2A=="], + "lru-cache": ["lru-cache@11.5.2", "", {}, "sha512-4pfM1Ff0x50o0tQwb5ucw/RzNyD0/YJME6IVcStalZuMWxdt3sR3huStTtxz4PUmvZfRguvDejasvQ2kifR11g=="], "lucide-react": ["lucide-react@1.24.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, "sha512-YT6mBD8lGKkg4nM39enlm94/sfJIiW0YKUT60fBy4YK8tai31ylg1VhGNWxkpSKHo9UagfnZqwIff3HTDQwXeA=="], @@ -1485,7 +1526,7 @@ "typed-query-selector": ["typed-query-selector@2.12.2", "", {}, "sha512-EOPFbyIub4ngnEdqi2yOcNeDLaX/0jcE1JoAXQDDMIthap7FoN795lc/SHfIq2d416VufXpM8z/lD+WRm2gfOQ=="], - "typescript": ["typescript@6.0.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw=="], + "typescript": ["typescript@7.0.2", "", { "optionalDependencies": { "@typescript/typescript-aix-ppc64": "7.0.2", "@typescript/typescript-darwin-arm64": "7.0.2", "@typescript/typescript-darwin-x64": "7.0.2", "@typescript/typescript-freebsd-arm64": "7.0.2", "@typescript/typescript-freebsd-x64": "7.0.2", "@typescript/typescript-linux-arm": "7.0.2", "@typescript/typescript-linux-arm64": "7.0.2", "@typescript/typescript-linux-loong64": "7.0.2", "@typescript/typescript-linux-mips64el": "7.0.2", "@typescript/typescript-linux-ppc64": "7.0.2", "@typescript/typescript-linux-riscv64": "7.0.2", "@typescript/typescript-linux-s390x": "7.0.2", "@typescript/typescript-linux-x64": "7.0.2", "@typescript/typescript-netbsd-arm64": "7.0.2", "@typescript/typescript-netbsd-x64": "7.0.2", "@typescript/typescript-openbsd-arm64": "7.0.2", "@typescript/typescript-openbsd-x64": "7.0.2", "@typescript/typescript-sunos-x64": "7.0.2", "@typescript/typescript-win32-arm64": "7.0.2", "@typescript/typescript-win32-x64": "7.0.2" }, "bin": { "tsc": "bin/tsc" } }, "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA=="], "uglify-js": ["uglify-js@3.19.3", "", { "bin": { "uglifyjs": "bin/uglifyjs" } }, "sha512-v3Xu+yuwBXisp6QYTcH4UbH+xYJXqnq2m/LtQVWKWzYc1iehYnLixoQDN9FH6/j9/oybfd6W9Ghwkl8+UMKTKQ=="], @@ -1551,28 +1592,6 @@ "@isaacs/fs-minipass/minipass": ["minipass@7.1.3", "", {}, "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A=="], - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="], - - "@opentelemetry/otlp-exporter-base/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/otlp-transformer/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="], - - "@opentelemetry/sdk-logs/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/sdk-logs/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - - "@opentelemetry/sdk-metrics/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="], - - "@opentelemetry/sdk-metrics/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="], - "@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-vgj7R3y3Wgx24IQaGPA/R6YFXLHVMOZ0uVEyIQPaWs+rd1AzfEMXlAC22FYwO1XkKR6NPsq7mUandH8oIRdZFw=="], "@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.2", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-TC8MkTuZUtcTSiFeuC0ksCh9QIJ5+F21MvZ4Wn4ORfYaFJ/0dsiudv5tVkejgwZlwQ39jL9WWDe2lz8x0WglOA=="], diff --git a/package.json b/package.json index 31b1e8499..da161644f 100644 --- a/package.json +++ b/package.json @@ -5,8 +5,9 @@ "type": "module", "packageManager": "bun@1.3.14", "patchedDependencies": { - "@ark/schema@0.56.1": "patches/@ark%2Fschema@0.56.1.patch", - "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch" + "@ark/schema@0.56.2": "patches/@ark%2Fschema@0.56.2.patch", + "puppeteer-core@25.3.0": "patches/puppeteer-core@25.3.0.patch", + "@agentclientprotocol/sdk@1.2.1": "patches/@agentclientprotocol%2Fsdk@1.2.1.patch" }, "workspaces": { "packages": [ @@ -14,17 +15,17 @@ "python/robomp/web" ], "catalog": { - "@agentclientprotocol/sdk": "0.25.0", + "@agentclientprotocol/sdk": "1.2.1", "@babel/generator": "^7.29.7", "@babel/parser": "^7.29.7", "@babel/traverse": "^7.29.7", "@babel/types": "^7.29.7", - "@biomejs/biome": "^2.5", - "@bufbuild/protobuf": "^2.12.0", - "@bufbuild/protoc-gen-es": "^2.12.0", + "@biomejs/biome": "^2.5.3", + "@bufbuild/protobuf": "^2.12.1", + "@bufbuild/protoc-gen-es": "^2.12.1", "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", - "@napi-rs/cli": "3.7.0", + "@napi-rs/cli": "3.7.2", "@oh-my-pi/hashline": "16.4.8", "@oh-my-pi/omp-stats": "16.4.8", "@oh-my-pi/pi-agent-core": "16.4.8", @@ -39,54 +40,54 @@ "@oh-my-pi/snapcompact": "16.4.8", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", - "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", + "@opentelemetry/exporter-trace-otlp-proto": "^0.220.0", "@opentelemetry/resources": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", "@opentelemetry/sdk-trace-node": "^2.7.1", - "@puppeteer/browsers": "^3.0.4", - "@tailwindcss/node": "^4.3.0", - "@tailwindcss/vite": "^4.3.0", + "@puppeteer/browsers": "^3.0.6", + "@tailwindcss/node": "^4.3.2", + "@tailwindcss/vite": "^4.3.2", "@types/babel__generator": "^7.27.0", "@types/babel__traverse": "^7.28.0", "@types/bun": "^1.3.14", "@types/react": "^19.2.17", "@types/react-dom": "^19.2.3", "@types/turndown": "5.0.6", - "@typescript/native-preview": "7.0.0-dev.20260609.1", + "@typescript/native-preview": "7.0.0-dev.20260707.2", "@xterm/headless": "^6.0.0", - "arktype": "2.2.2", + "arktype": "2.2.3", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", "diff": "^9.0.0", "fflate": "0.8.3", "fastembed": "2.1.0", - "fast-xml-parser": "^5.9.0", + "fast-xml-parser": "^5.9.3", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", "header-generator": "^2.1.82", - "linkedom": "^0.18.12", - "lint-staged": "^17.0.7", - "lru-cache": "11.5.1", - "lucide-react": "^1.17.0", + "linkedom": "^0.18.13", + "lint-staged": "^17.0.8", + "lru-cache": "11.5.2", + "lucide-react": "^1.24.0", "mammoth": "^1.12.0", - "marked": "^18.0.5", - "mupdf": "^1.27.0", + "marked": "^18.0.6", + "mupdf": "^1.28.0", "onnxruntime-node": "1.26.0", - "postcss": "^8.5.15", - "prettier": "^3.8.4", + "postcss": "^8.5.16", + "prettier": "^3.9.5", "puppeteer-core": "25.3.0", "react": "19.2.7", "react-chartjs-2": "^5.3.1", "react-dom": "19.2.7", "regexp-tree": "^0.1.27", - "solid-js": "^1.9.13", - "tailwindcss": "^4.3.0", + "solid-js": "^1.9.14", + "tailwindcss": "^4.3.2", "ts-morph": "^28.0.0", "turndown": "7.2.4", "turndown-plugin-gfm": "1.0.2", - "typescript": "^6.0.3", - "vite": "^8.0.16", + "typescript": "^7.0.2", + "vite": "^8.1.4", "vite-plugin-solid": "^2.11.12", "winston": "^3.19.0", "winston-daily-rotate-file": "^5.0.0", @@ -94,7 +95,7 @@ } }, "overrides": { - "@ark/schema": "0.56.1" + "@ark/schema": "0.56.2" }, "scripts": { "setup": "bun install && bun run build:native && bun --cwd=packages/coding-agent link && sh scripts/link-omp.sh", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8a90cde6c..4836e57dc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -25,6 +25,7 @@ - Added visual markers in the transcript for elided tool calls that have no corresponding result - Updated status event log to prioritize the most recent entries in the display window - Updated the snapcompact shape preview transcript to use the compact scope format shown to models during compaction. +- Bumped `@agentclientprotocol/sdk` 0.25.0 → 1.2.1 (major); patched the package's `exports` map to restore the `dist/schema/zod.gen.js` subpath the SDK no longer publishes, which our tests import for response-shape validation. ### Removed diff --git a/patches/@agentclientprotocol%2Fsdk@1.2.1.patch b/patches/@agentclientprotocol%2Fsdk@1.2.1.patch new file mode 100644 index 000000000..7e779ed74 --- /dev/null +++ b/patches/@agentclientprotocol%2Fsdk@1.2.1.patch @@ -0,0 +1,18 @@ +diff --git a/package.json b/package.json +index 7684b5183562eaa3f585447db5b3a0e2a28575fc..09ce449ba97c65f3afe2422435aa626d362ef44d 100644 +--- a/package.json ++++ b/package.json +@@ -50,7 +50,12 @@ + "import": "./dist/node-adapter.js", + "default": "./dist/node-adapter.js" + }, +- "./schema/schema.json": "./schema/schema.json" ++ "./schema/schema.json": "./schema/schema.json", ++ "./dist/schema/zod.gen.js": { ++ "types": "./dist/schema/zod.gen.d.ts", ++ "import": "./dist/schema/zod.gen.js", ++ "default": "./dist/schema/zod.gen.js" ++ } + }, + "directories": { + "example": "examples" diff --git a/patches/@ark%2Fschema@0.56.1.patch b/patches/@ark%2Fschema@0.56.2.patch similarity index 100% rename from patches/@ark%2Fschema@0.56.1.patch rename to patches/@ark%2Fschema@0.56.2.patch From c69c04836119c8e6ce9fa74b06ea8e9f85610dd7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:51:34 +0200 Subject: [PATCH 16/28] ci: stabilized test runs by resetting settings and shrinking UI/TUI chunks - Initialized in-memory settings in the repro issue test by resetting settings before theme setup. - Reduced the UI/TUI CI test bucket chunk size from 10 to 5 to avoid cumulative Bun GC heap aborts. --- .../repro-issue-1955-sendmessage-double-render.test.ts | 4 ++++ scripts/ci-test-ts.ts | 10 +++++++++- 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts b/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts index 239790ed8..fbc75a45b 100644 --- a/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts +++ b/packages/coding-agent/test/repro-issue-1955-sendmessage-double-render.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeAll, describe, expect, test, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ExtensionActions, ExtensionCommandContextActions, @@ -31,6 +32,9 @@ import { Container } from "@oh-my-pi/pi-tui"; * leaving two identical custom-message components in the chat. */ beforeAll(async () => { + // renderInitialMessages reads the global Settings (display.collapseCompacted). + resetSettingsForTest(); + await Settings.init({ inMemory: true }); await initTheme(); }); diff --git a/scripts/ci-test-ts.ts b/scripts/ci-test-ts.ts index e43a75d12..ac4ff2736 100755 --- a/scripts/ci-test-ts.ts +++ b/scripts/ci-test-ts.ts @@ -64,9 +64,17 @@ const validModes: Record = { // under the CI runner's OOM ceiling (a single 170–370-file invocation gets // SIGKILLed at 137). The singleton/global-state bucket is left whole: its suites // co-locate in one process to exercise process-wide state, so they must not split. +// +// The UI/TUI bucket uses a smaller chunk (5) than the others: its suites build up +// native ghostty-vt cells, and bun 1.3.14's GC aborts (SIGTRAP/SIGABRT, exit +// 133/134 inside DOMGCOutputConstraint marking) once ~10 such files share a heap, +// even with the GC-marker knobs below. Bisection showed no single file is at +// fault — the crash is cumulative heap volume. Under a 256MB-forced heap, a +// 10-file chunk aborts ~50% of runs while either 5-file half is 0/20; halving the +// chunk keeps each process under the threshold. const codingAgentBucketPlans: Record = { singleton: { label: "singleton/global-state bucket", parallel: 1 }, - ui: { label: "UI/TUI bucket", parallel: 1, chunkSize: 10 }, + ui: { label: "UI/TUI bucket", parallel: 1, chunkSize: 5 }, runtime: { label: "runtime/session bucket", parallel: 1, chunkSize: 10 }, native: { label: "native/tooling/browser/unit bucket", parallel: 1, chunkSize: 10 }, }; From 59ecd2a4d8b488ffca1c702084f1999fd65e33d2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 18:53:19 +0200 Subject: [PATCH 17/28] refactor(coding-agent/modes): added ACP type guards for elicitation narrowing - Added `isAcceptedElicitation` in the ACP agent to narrow accepted elicitations before accessing response content. - Added `isFormElicitation` in ACP tests and used it to narrow form-mode requests before assertions. --- packages/coding-agent/src/modes/acp/acp-agent.ts | 9 ++++++++- packages/coding-agent/test/acp-agent.test.ts | 15 +++++++++++---- 2 files changed, 19 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 5b31ec935..3bca86ddc 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -351,12 +351,19 @@ async function elicitFromAcpClient( finish(undefined); }); const response = await promise; - if (response?.action !== "accept" || !response.content) { + if (!isAcceptedElicitation(response) || !response.content) { return undefined; } return response.content.value; } +/** Narrows a `CreateElicitationResponse` to the accepted-with-content branch; the SDK's `action: string` catch-all arm otherwise defeats literal narrowing on `action !== "accept"`. */ +function isAcceptedElicitation( + response: CreateElicitationResponse | undefined, +): response is Extract { + return response?.action === "accept"; +} + /** * Build an {@link ExtensionUIContext} that translates skill/extension UI * requests into ACP elicitations against `connection` for the session diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index f8892c557..886fedb6c 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -2244,6 +2244,13 @@ describe("ACP agent", () => { return { connection, calls }; } + /** Narrows `CreateElicitationRequest` to the `mode: "form"` branch; the SDK's `mode: string` catch-all arm otherwise defeats literal narrowing on `mode !== "form"`. */ + function isFormElicitation( + request: CreateElicitationRequest, + ): request is Extract { + return request.mode === "form"; + } + it("translates select to a single-property string-enum elicitation", async () => { const { connection, calls } = createElicitConnection(async () => ({ action: "accept", @@ -2258,7 +2265,7 @@ describe("ACP agent", () => { const request = calls[0]!; expect(request.mode).toBe("form"); expect(request.message).toBe("Pick one"); - if (request.mode !== "form" || !("sessionId" in request)) { + if (!isFormElicitation(request) || !("sessionId" in request)) { throw new Error("expected session-scoped form elicitation"); } expect(request.sessionId).toBe("session-select"); @@ -2281,7 +2288,7 @@ describe("ACP agent", () => { expect(result).toBe(true); expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") { + if (!isFormElicitation(request)) { throw new Error("expected form-mode elicitation"); } expect(request.message).toBe("Proceed?\n\nThis will overwrite the file."); @@ -2301,7 +2308,7 @@ describe("ACP agent", () => { expect(result).toBe("claude"); expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") { + if (!isFormElicitation(request)) { throw new Error("expected form-mode elicitation"); } expect(request.message).toBe("Your name?"); @@ -2450,7 +2457,7 @@ describe("ACP agent", () => { expect(calls).toHaveLength(1); const request = calls[0]!; - if (request.mode !== "form") throw new Error("expected form-mode elicitation"); + if (!isFormElicitation(request)) throw new Error("expected form-mode elicitation"); expect(request.requestedSchema.properties?.value).toEqual({ type: "string" }); }); From 896c4bb17b70d9592e11d4a8617474bac773f255 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 19:00:26 +0200 Subject: [PATCH 18/28] fix(coding-agent/tools): capped expanded streaming diff previews to a viewport-sized tail - Bounded expanded partial edit diff rendering in `formatStreamingDiff` to `previewWindowRows()` instead of an unbounded budget, preventing runaway preview growth during live updates. - Updated streaming diff tests to simulate terminal height and verify expanded previews stay full only within the viewport, then switch to a truncated tail with the `more lines above` marker when too tall. - Reinitialized in-memory `Settings` before each initial-messages test since the test suite reads global display configuration and needs isolation. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/edit/renderer.ts | 23 +++--- .../utils/render-initial-messages.test.ts | 9 ++- .../test/tools/edit-renderer.test.ts | 70 ++++++++++++------- 4 files changed, 67 insertions(+), 36 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4836e57dc..7cc781bcd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -34,6 +34,7 @@ ### Fixed +- Fixed expanded (ctrl+O) streaming edit previews duplicating the tool box in terminal scrollback: an unbounded live diff scrolled above the native-scrollback commit boundary mid-stream, freezing a stale preview snapshot that the finalized render then recommitted below. Expanded previews now use a viewport-sized tail window; the full diff still renders once the result finalizes. - Fixed quadratic growth of `--mode json` logs by eliding redundant message snapshots and payloads - Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache - Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 14682a59e..cd11ab3c5 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -423,22 +423,25 @@ function formatStreamingDiff( cache?: RenderedStringCache, ): string { if (!diff) return ""; - // Clamp the collapsed tail to the viewport so a tall or fast-growing diff - // cannot outgrow the live window. Otherwise its mutating tail scrolls above - // the native-scrollback commit boundary and the engine re-commits a fresh - // snapshot every streamed frame, stacking duplicate "… more lines above" - // previews in history. The budget is VISUAL rows (a long wrapped line counts + // Clamp the tail to the viewport so a tall or fast-growing diff cannot + // outgrow the live window. Otherwise its mutating rows scroll above the + // native-scrollback commit boundary mid-stream and freeze into immutable + // history as a stale preview snapshot; the finalize repair then recommits + // the final render below it — a duplicated block on the tape. Collapsed + // gets a short fixed tail; expanded widens it to the viewport-sized window, + // never unbounded. The budget is VISUAL rows (a long wrapped line counts // for more than one) at the framed block's inner width (border only — - // contentPaddingLeft is 0); only the visible suffix is syntax-colored, so the - // cheap raw-line wrap walk keeps the per-chunk cost bounded. innerWidth/budget - // are in the cache salt so a resize re-slices. + // contentPaddingLeft is 0); only the visible suffix is syntax-colored, so + // the cheap raw-line wrap walk keeps the per-chunk cost bounded. + // innerWidth/budget are in the cache salt so a resize re-slices. const innerWidth = Math.max(1, width - 2); - const budget = expanded ? Number.POSITIVE_INFINITY : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); + const budget = expanded ? previewWindowRows() : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${innerWidth}:${budget}`, diff, () => { // "Cursor" tail window: pin the last rows to the bottom so freshly streamed // changes stay on screen. The whole-file diff is recomputed every chunk and // its Myers alignment is not monotonic in payload length, so a hunk-aware - // window stutters as rows move between hunks. Expanded lifts the cap. + // window stutters as rows move between hunks. Expanded widens the window + // to the viewport; the full diff appears once the result finalizes. const allLines = diff.replace(/\n+$/u, "").split("\n"); let visualUsed = 0; let cut = allLines.length; diff --git a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts index 03f64ac3d..bc3df339f 100644 --- a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts +++ b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts @@ -11,7 +11,7 @@ * scrollback-clearing repaint (`clearTerminalHistory`). */ -import { afterEach, beforeAll, describe, expect, it, type Mock, vi } from "bun:test"; +import { afterEach, beforeAll, beforeEach, describe, expect, it, type Mock, vi } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, ImageContent, Usage } from "@oh-my-pi/pi-ai"; import { kStreamingPartialJson } from "@oh-my-pi/pi-ai/utils/block-symbols"; @@ -28,6 +28,13 @@ beforeAll(() => { initTheme(); }); +beforeEach(async () => { + // afterEach resets Settings, but renderInitialMessages reads the global + // Settings (display.collapseCompacted) — re-init before every test. + resetSettingsForTest(); + await Settings.init({ inMemory: true }); +}); + const originalImageProtocol = TERMINAL.imageProtocol; afterEach(() => { diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index 7b84723cf..7cee7fcb6 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -57,34 +57,54 @@ describe("editToolRenderer", () => { expect(rendered).toContain("packages/coding-agent/src/edit/renderer.ts"); }); - it("lifts the streaming diff tail window when expanded", async () => { + it("windows the expanded streaming diff to the viewport tail", async () => { const uiTheme = await getUiTheme(); - const diff = Array.from({ length: 20 }, (_, index) => - index === 0 ? "-head-line-1" : `+tail-line-${index + 1}`, - ).join("\n"); - const renderPreview = (expanded: boolean): string => - Bun.stripANSI( - editToolRenderer - .renderCall( - { file_path: "/tmp/preview.ts", previewDiff: diff }, - { expanded, isPartial: true, spinnerFrame: 0, renderContext: { editMode: "replace" } }, - uiTheme, - ) - .render(200) - .join("\n"), - ); + // Pin a tall viewport so previewWindowRows() (rows - reserve) lands at 30: + // collapsed stays at the 12-row fixed tail, expanded widens to 30. + const originalRowsDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "rows"); + Object.defineProperty(process.stdout, "rows", { value: 50, configurable: true }); + try { + const makeDiff = (length: number): string => + Array.from({ length }, (_, index) => (index === 0 ? "-head-line-1" : `+tail-line-${index + 1}`)).join("\n"); + const renderPreview = (diff: string, expanded: boolean): string => + Bun.stripANSI( + editToolRenderer + .renderCall( + { file_path: "/tmp/preview.ts", previewDiff: diff }, + { expanded, isPartial: true, spinnerFrame: 0, renderContext: { editMode: "replace" } }, + uiTheme, + ) + .render(200) + .join("\n"), + ); - const collapsed = renderPreview(false); - expect(collapsed).toContain("tail-line-20"); - expect(collapsed).not.toContain("head-line-1"); - expect(collapsed).toContain("more lines above"); - expect(collapsed).toContain("(preview)"); + const collapsed = renderPreview(makeDiff(20), false); + expect(collapsed).toContain("tail-line-20"); + expect(collapsed).not.toContain("head-line-1"); + expect(collapsed).toContain("more lines above"); + expect(collapsed).toContain("(preview)"); - const expanded = renderPreview(true); - expect(expanded).toContain("head-line-1"); - expect(expanded).toContain("tail-line-20"); - expect(expanded).not.toContain("more lines above"); - expect(expanded).not.toContain("(preview)"); + // Within the viewport window, expanded shows the whole diff. + const expanded = renderPreview(makeDiff(20), true); + expect(expanded).toContain("head-line-1"); + expect(expanded).toContain("tail-line-20"); + expect(expanded).not.toContain("more lines above"); + expect(expanded).not.toContain("(preview)"); + + // Beyond it, expanded stays a viewport-sized tail window: an unbounded + // live preview scrolls above the native-scrollback commit boundary and + // freezes a stale snapshot that duplicates the block at finalize. + const expandedTall = renderPreview(makeDiff(40), true); + expect(expandedTall).toContain("tail-line-40"); + expect(expandedTall).not.toContain("head-line-1"); + expect(expandedTall).toContain("more lines above"); + } finally { + if (originalRowsDescriptor) { + Object.defineProperty(process.stdout, "rows", originalRowsDescriptor); + } else { + Reflect.deleteProperty(process.stdout, "rows"); + } + } }); it("uses hashline input headers for streaming call path without apply_patch errors", async () => { From f80fb4836f0194245776e47a50fb3dc1f9a542b6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 19:13:00 +0200 Subject: [PATCH 19/28] feat: updated prewalk-plan guidance and added hot-reload marker - Refined the coding-agent prewalk-plan system prompt to require 5-9 meaningful todo steps with concrete targets and verification, and removed the requirement for a super-detailed single-reply plan format. - Added a `hot-reload-probe` marker comment at the end of Harbor Manager's `server.ts`. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/prompts/system/prewalk-plan.md | 2 +- packages/harbor-manager/src/server.ts | 2 ++ 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7cc781bcd..43f0267f6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -19,7 +19,7 @@ ### Changed - Updated `--mode json` logs to include provider payloads in auto-compaction events -- Refined the prewalk planning instructions to require a super-detailed todo list with one item per concrete step, each specifying its target and verification method +- Refined prewalk planning to require 5-9 meaningful todo items with concrete targets and checks - Updated tangential agent forks to ignore parent session history and focus exclusively on the new request - Hardened `/tan` fork isolation: the clone's inherited todo list is cleared at fork (parent todo reminders no longer drag the tan back onto the parent's task), the fork notice warns that the parent is concurrently editing the same working directory, and the notice is re-injected after each compaction so the fork boundary survives summarization - Added visual markers in the transcript for elided tool calls that have no corresponding result diff --git a/packages/coding-agent/src/prompts/system/prewalk-plan.md b/packages/coding-agent/src/prompts/system/prewalk-plan.md index 1a2d4b74c..3f4f63e9e 100644 --- a/packages/coding-agent/src/prompts/system/prewalk-plan.md +++ b/packages/coding-agent/src/prompts/system/prewalk-plan.md @@ -8,6 +8,6 @@ First, state the plan itself, explicitly and comprehensively: Be thorough and concrete — this plan is the reference for the remainder of the run. You may verify details with tools after the plan is written, never before. -Then, only once the plan above is complete and detailed, in the SAME reply, capture it as a SUPER-DETAILED todo list (the todo tool): one item per concrete step from the plan — each naming its exact file/symbol/command target and its verification — not a handful of vague phase headings. The todo list must be precise enough that every item can be checked off against an observable result. +Then, only once the plan above is complete, in the SAME reply, capture it as a todo list (the todo tool): 5-9 items, one per MEANINGFUL step, each naming its concrete target and its verification. Only steps that change or verify code belong on the list — no reporting, bookkeeping, cleanup-ceremony, or release-note items. The todo list serves the task, never the reverse: when reality disagrees with an item, fix the actual problem rather than working the checklist. This is a checkpoint, not a final answer: do not end your turn on the plan alone — after recording the todo list, continue the task; do not stop here. diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index e9af7650b..39ecb7447 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -645,3 +645,5 @@ if (import.meta.main) { process.on("SIGINT", shutdown); process.on("SIGTERM", shutdown); } + +// hot-reload-probe From 4c1c5f40d834b1b22fe04bfb37994b903ed52895 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 19:18:17 +0200 Subject: [PATCH 20/28] feat(launch): rendered launch logs from daemon terminal byte streams - Changed daemon log reads to return both sanitized display text and a raw `terminalText` slice, and included it on log RPC responses for PTY runs when grep was not used. - Extended the logs result contract and launch tool rendering to consume `terminalText`, reconstruct terminal output, and display it in framed, preview-capped sections. - Kept terminal row layout stable by writing space characters for empty cells when reading rows, preserving spacing during output reconstruction. --- .../src/cli/gallery-fixtures/shell.ts | 29 ++++++++----- packages/coding-agent/src/launch/broker.ts | 41 +++++++++++++------ packages/coding-agent/src/launch/protocol.ts | 4 ++ packages/coding-agent/src/tools/bash.ts | 10 ++++- packages/coding-agent/src/tools/launch.ts | 39 ++++++++++++++---- .../coding-agent/src/tools/render-utils.ts | 3 ++ .../coding-agent/src/tools/terminal-output.ts | 6 +-- .../test/tools/launch-renderer.test.ts | 19 ++++++--- .../coding-agent/test/tools/launch.test.ts | 8 +++- 9 files changed, 115 insertions(+), 44 deletions(-) diff --git a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts index d734a7da3..1261b672c 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts @@ -102,24 +102,33 @@ export const shellFixtures: Record = { launch_logs: { label: "Launch", renderer: "launch", - args: { op: "logs", name: "web", lines: 100, follow: true, cursor: 1842, timeout: 30 }, + args: { op: "logs", name: "comp-debug", lines: 100, follow: true, cursor: 233_512, timeout: 30 }, result: { content: [ { type: "text", text: [ - "$ bun run dev", - " VITE v6.0.3 ready in 312 ms", - "", - " ➜ Local: http://localhost:5173/", - " ➜ Network: use --host to expose", - "12:04:11 [vite] hmr update /src/App.tsx", - "12:04:15 [vite] hmr update /src/components/Chart.tsx", - "[web: running; cursor=2210]", + "Breakpoint 1: 3 locations.", + "(lldb) run", + "Process 726 launched: '/tmp/compiler'", + "frame #0: 0x0000000100012f80 compiler`parse_expression", + "[comp-debug: ready; cursor=233797]", ].join("\n"), }, ], - details: { op: "logs", cursor: 2210, timedOut: false, state: "running" }, + details: { + op: "logs", + cursor: 233_797, + timedOut: false, + state: "ready", + terminalRows: [ + "\x1b[0mBreakpoint 1: 3 locations.", + "\x1b[0m(lldb) run", + "\x1b[0mProcess 726 launched: '/tmp/compiler'", + "\x1b[0mframe #0: 0x0000000100012f80 compiler`parse_expression", + "\x1b[0m\x1b[1;38;5;2m(lldb)\x1b[0m ", + ], + }, }, errorResult: { content: [{ type: "text", text: "No daemon named web" }], diff --git a/packages/coding-agent/src/launch/broker.ts b/packages/coding-agent/src/launch/broker.ts index 9c8aa5db2..9511ae9a9 100644 --- a/packages/coding-agent/src/launch/broker.ts +++ b/packages/coding-agent/src/launch/broker.ts @@ -4,7 +4,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { Process, type PtyRunResult, PtySession } from "@oh-my-pi/pi-natives"; import { isEexist, isEnoent, logger, postmortem, sanitizeText } from "@oh-my-pi/pi-utils"; -import { truncateHead, truncateTail } from "../session/streaming-output"; +import { truncateHead, truncateHeadBytes, truncateTail, truncateTailBytes } from "../session/streaming-output"; import { workerEnvFromParent } from "../subprocess/worker-client"; import { daemonBrokerEndpoint } from "./paths"; import { hasLiveDaemonProjectPresence } from "./presence"; @@ -76,6 +76,11 @@ interface BrokerLease { instanceId: string; } +interface DaemonLogRead { + text: string; + terminalText: string; +} + function quoteShellArg(value: string): string { return `'${value.replaceAll("'", `'\\''`)}'`; } @@ -155,7 +160,7 @@ class DaemonLog { return text; } - async read(head: boolean, lines: number, grep?: string): Promise { + async read(head: boolean, lines: number, grep?: string): Promise { await this.#queue; await this.#writer.flush(); return DaemonLog.readFiles(this.#path, this.#previousPath, head, lines, grep); @@ -168,19 +173,19 @@ class DaemonLog { await this.#writer.end(); } - static async readDir(dir: string, head: boolean, lines: number, grep?: string): Promise { - return DaemonLog.readFiles(path.join(dir, LOG_FILE), path.join(dir, PREVIOUS_LOG_FILE), head, lines, grep); - } - static async readFiles( logPath: string, previousPath: string, head: boolean, lines: number, grep?: string, - ): Promise { + ): Promise { const [previous, current] = await Promise.all([fileTextSlice(previousPath, head), fileTextSlice(logPath, head)]); - let text = `${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`; + const combined = `${previous}${previous && current && !previous.endsWith("\n") ? "\n" : ""}${current}`; + const terminalText = head + ? truncateHeadBytes(combined, LOG_READ_BYTES).text + : truncateTailBytes(combined, LOG_READ_BYTES).text; + let text = sanitizeText(terminalText); if (grep) { let pattern: RegExp; try { @@ -190,11 +195,14 @@ class DaemonLog { } text = text .split("\n") - .filter(line => pattern.test(sanitizeText(line))) + .filter(line => pattern.test(line)) .join("\n"); } const options = { maxLines: lines, maxBytes: 256 * 1024 }; - return head ? truncateHead(text, options).content : truncateTail(text, options).content; + return { + text: head ? truncateHead(text, options).content : truncateTail(text, options).content, + terminalText, + }; } async #rotate(): Promise { @@ -755,13 +763,20 @@ class DaemonBroker { timedOut = !changed; } const lines = Math.max(1, Math.min(1_000, Math.floor(operation.lines))); - const text = record.log + const output = record.log ? await record.log.read(operation.head, lines, operation.grep) - : await DaemonLog.readDir(record.dir, operation.head, lines, operation.grep); + : await DaemonLog.readFiles( + path.join(record.dir, LOG_FILE), + path.join(record.dir, PREVIOUS_LOG_FILE), + operation.head, + lines, + operation.grep, + ); return { op: "logs", name: record.snapshot.name, - text, + text: output.text, + terminalText: record.spec.pty && operation.grep === undefined ? output.terminalText : undefined, cursor: record.snapshot.outputBytes, timedOut, state: record.snapshot.state, diff --git a/packages/coding-agent/src/launch/protocol.ts b/packages/coding-agent/src/launch/protocol.ts index 0ad3e31cf..5ea71b2a6 100644 --- a/packages/coding-agent/src/launch/protocol.ts +++ b/packages/coding-agent/src/launch/protocol.ts @@ -101,6 +101,8 @@ export type DaemonRpcResult = op: "logs"; name: string; text: string; + /** Raw PTY byte stream used only to reconstruct the terminal screen. */ + terminalText?: string; cursor: number; timedOut: boolean; state: DaemonState; @@ -353,6 +355,8 @@ export function parseDaemonRpcResult(operation: DaemonOperation, value: unknown) op: "logs", name: stringValue(source.name, "result.name"), text: typeof source.text === "string" ? source.text : "", + terminalText: + source.terminalText === undefined ? undefined : rawString(source.terminalText, "result.terminalText"), cursor: numberValue(source.cursor, "result.cursor"), timedOut: booleanValue(source.timedOut, "result.timedOut"), state: daemonState(source.state), diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index e272c230c..ccfea33c1 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -36,12 +36,18 @@ import { stripRawOutputArtifactNotice, } from "./output-meta"; import { resolveToCwd } from "./path-utils"; -import { capPreviewLines, formatToolWorkingDirectory, previewWindowRows, replaceTabs } from "./render-utils"; +import { + capPreviewLines, + DEFAULT_TERMINAL_PREVIEW_LINES, + formatToolWorkingDirectory, + previewWindowRows, + replaceTabs, +} from "./render-utils"; import { ToolAbortError, ToolError } from "./tool-errors"; import { toolResult } from "./tool-result"; import { clampTimeout, TOOL_TIMEOUTS } from "./tool-timeouts"; -export const BASH_DEFAULT_PREVIEW_LINES = 10; +export const BASH_DEFAULT_PREVIEW_LINES = DEFAULT_TERMINAL_PREVIEW_LINES; const BASH_ENV_NAME_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/; const DEFAULT_AUTO_BACKGROUND_THRESHOLD_MS = 60_000; diff --git a/packages/coding-agent/src/tools/launch.ts b/packages/coding-agent/src/tools/launch.ts index 31c628bbf..2c68ae446 100644 --- a/packages/coding-agent/src/tools/launch.ts +++ b/packages/coding-agent/src/tools/launch.ts @@ -16,12 +16,13 @@ import type { DaemonOperation, DaemonRpcResult, DaemonSnapshot, DaemonSpec, Daem import { renderTerminalOutput } from "../launch/terminal-output"; import type { Theme, ThemeColor } from "../modes/theme/theme"; import launchDescription from "../prompts/tools/launch.md" with { type: "text" }; -import { renderStatusLine } from "../tui"; +import { framedBlock, outputBlockContentWidth, renderStatusLine } from "../tui"; import type { ToolSession } from "."; import { resolveToCwd } from "./path-utils"; import { capPreviewLines, createCachedComponent, + DEFAULT_TERMINAL_PREVIEW_LINES, formatDuration, formatExpandHint, formatMoreItems, @@ -283,10 +284,13 @@ async function toolDetails(result: DaemonRpcResult, params: LaunchParams): Promi case "list": return { op: "list", daemons: result.daemons }; case "logs": { - const terminalRows = await renderTerminalOutput(result.text, { - head: params.head ?? false, - maxRows: Math.min(1_000, Math.floor(params.lines ?? 100)), - }); + const terminalRows = + result.terminalText === undefined + ? undefined + : await renderTerminalOutput(result.terminalText, { + head: params.head ?? false, + maxRows: Math.min(1_000, Math.floor(params.lines ?? 100)), + }); return { op: "logs", cursor: result.cursor, @@ -600,13 +604,32 @@ export const launchToolRenderer = { theme, ); + if (op === "logs") { + return framedBlock(theme, width => { + const innerWidth = outputBlockContentWidth(width); + const rows = body.map(line => truncateToWidth(line, innerWidth)); + return { + header, + state: options.isPartial ? "pending" : failed ? "error" : "success", + sections: [ + { + label: theme.fg("toolTitle", "Output"), + lines: capPreviewLines(rows, theme, { + expanded: options.expanded, + max: DEFAULT_TERMINAL_PREVIEW_LINES, + }), + }, + ], + width, + }; + }); + } + return createCachedComponent( () => options.expanded, (width, expanded) => { let visible = body; - if (op === "logs") { - visible = capPreviewLines(body, theme, { expanded }); - } else if (!expanded && op === "list" && body.length > PREVIEW_LIMITS.COLLAPSED_ITEMS) { + if (!expanded && op === "list" && body.length > PREVIEW_LIMITS.COLLAPSED_ITEMS) { const remaining = body.length - PREVIEW_LIMITS.COLLAPSED_ITEMS; visible = [ ...body.slice(0, PREVIEW_LIMITS.COLLAPSED_ITEMS), diff --git a/packages/coding-agent/src/tools/render-utils.ts b/packages/coding-agent/src/tools/render-utils.ts index 12228a1cb..9a6e857af 100644 --- a/packages/coding-agent/src/tools/render-utils.ts +++ b/packages/coding-agent/src/tools/render-utils.ts @@ -62,6 +62,9 @@ export const PREVIEW_LIMITS = { DIFF_COLLAPSED_LINES: 40, } as const; +/** Default number of terminal output rows shown before expansion. */ +export const DEFAULT_TERMINAL_PREVIEW_LINES = 10; + /** Truncation lengths for different content types */ export const TRUNCATE_LENGTHS = { /** Short titles, labels */ diff --git a/packages/coding-agent/src/tools/terminal-output.ts b/packages/coding-agent/src/tools/terminal-output.ts index e77dc1f4e..6f1043ff4 100644 --- a/packages/coding-agent/src/tools/terminal-output.ts +++ b/packages/coding-agent/src/tools/terminal-output.ts @@ -114,10 +114,8 @@ export function readTerminalRows(terminal: XtermTerminal, startRow: number, rowC if (!cell) break; const chars = cell.getChars(); const width = Math.max(1, cell.getWidth()); - if (chars) { - cells.push({ chars, style: cellStyle(cell) }); - if (chars !== " ") lastContent = cells.length - 1; - } + cells.push({ chars: chars || " ", style: cellStyle(cell) }); + if (chars && chars !== " ") lastContent = cells.length - 1; column += width; } diff --git a/packages/coding-agent/test/tools/launch-renderer.test.ts b/packages/coding-agent/test/tools/launch-renderer.test.ts index 9dfb27f47..fea9522ce 100644 --- a/packages/coding-agent/test/tools/launch-renderer.test.ts +++ b/packages/coding-agent/test/tools/launch-renderer.test.ts @@ -79,17 +79,24 @@ describe("launchToolRenderer", () => { ); expect(rendered[0]).toContain("Launch logs"); expect(rendered[0]).toContain("cursor 2210"); - expect(rendered).toContain("line one"); - expect(rendered).toContain("line two"); + expect(rendered.some(line => line.includes("line one"))).toBe(true); + expect(rendered.some(line => line.includes("line two"))).toBe(true); expect(rendered.some(line => line.includes("[web: running"))).toBe(false); + expect(rendered[0]).toContain("╭"); + expect(rendered.some(line => line.includes("Output"))).toBe(true); + expect(rendered.at(-1)).toContain("╰"); }); it("replays terminal screen rows so cursor rewrites retain their final color and weight", async () => { - const terminalRows = await renderTerminalOutput("\x1b[1;31mold\x1b[0m\r\x1b[1;32mready\x1b[0m\x1b[K", { - head: false, - maxRows: 10, - }); + const terminalRows = await renderTerminalOutput( + "\x1b[1;31mold\x1b[0m\r\x1b[2K\x1b[12G\x1b[1;32mready\x1b[0m\x1b[K", + { + head: false, + maxRows: 10, + }, + ); if (terminalRows === undefined) throw new Error("terminal replay failed"); + expect(Bun.stripANSI(terminalRows[0] ?? "")).toBe(" ready"); const uiTheme = await theme(); const component = launchToolRenderer.renderResult( diff --git a/packages/coding-agent/test/tools/launch.test.ts b/packages/coding-agent/test/tools/launch.test.ts index 84b6bd48c..df0bb0bda 100644 --- a/packages/coding-agent/test/tools/launch.test.ts +++ b/packages/coding-agent/test/tools/launch.test.ts @@ -59,6 +59,8 @@ describe("daemon broker", () => { `process.stdin.setRawMode?.(true); process.stdin.setEncoding("utf8"); process.stdin.resume(); +process.stdout.write("\\x1b[2J\\x1b[H"); +for (let index = 0; index < 25; index++) process.stdout.write("BOOT:" + index + "\\n"); process.stdout.write("\\x1b[1;32mREADY\\x1b[0m\\n"); process.stdin.on("data", chunk => process.stdout.write("INPUT:" + JSON.stringify(chunk) + "\\n")); setInterval(() => {}, 1000); @@ -114,8 +116,12 @@ setInterval(() => {}, 1000); expect(logs.op).toBe("logs"); if (logs.op !== "logs") throw new Error("unexpected logs result"); expect(logs.text).toContain("READY"); - expect(logs.text).toContain("\x1b[1;32mREADY\x1b[0m"); + expect(logs.text).not.toContain("\x1b"); + expect(logs.text).not.toContain("BOOT:0"); expect(logs.text).toContain('INPUT:"run\\r"'); + expect(logs.terminalText).toContain("\x1b[2J\x1b[H"); + expect(logs.terminalText).toContain("\x1b[1;32mREADY\x1b[0m"); + expect(logs.terminalText).toContain("BOOT:0"); const stopped = await first.request({ op: "stop", name: "debugger", timeoutMs: 2_000 }); expect(stopped.op).toBe("stop"); From 35d3e49d1f5281d094beb63f501db397e828ea32 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 19:22:06 +0200 Subject: [PATCH 21/28] fix(harbor-manager): prevented dev-server teardown rejections from killing the manager - Added a process-wide `__harborManagerHooks` flag on `globalThis` so signal and unhandled-rejection handlers are registered only once during Bun `--hot` re-execution. - Added `isDevStreamTeardown` handling to ignore `ERR_STREAM_RELEASE_LOCK` unhandled rejections from dev stream teardown while rethrowing all other unhandled rejections. - Added `react-refresh` to the Harbor Manager package manifest and lockfile dependencies. --- bun.lock | 3 ++ packages/coding-agent/CHANGELOG.md | 2 +- packages/harbor-manager/package.json | 3 +- packages/harbor-manager/src/server.ts | 48 +++++++++++++++++++-------- 4 files changed, 41 insertions(+), 15 deletions(-) diff --git a/bun.lock b/bun.lock index 610d212c0..6e26f48e8 100644 --- a/bun.lock +++ b/bun.lock @@ -164,6 +164,7 @@ "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", "@types/react-dom": "^19.1.0", + "react-refresh": "^0.18.0", }, }, "packages/hashline": { @@ -1418,6 +1419,8 @@ "react-dom": ["react-dom@19.2.7", "", { "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { "react": "^19.2.7" } }, "sha512-t0BRVXvbiE/o20Hfw669rLbMCDWtYZLvmJigy2f0MxsXF+71pxhR3xOkspmsO8h3ZlNzyibAmtCa3l4lYKk6gQ=="], + "react-refresh": ["react-refresh@0.18.0", "", {}, "sha512-QgT5//D3jfjJb6Gsjxv0Slpj23ip+HtOpnNgnb2S5zU3CB26G/IDPGoy4RJB42wzFE46DRsstbW6tKHoKbhAxw=="], + "readable-stream": ["readable-stream@3.6.2", "", { "dependencies": { "inherits": "^2.0.3", "string_decoder": "^1.1.1", "util-deprecate": "^1.0.1" } }, "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA=="], "regexp-tree": ["regexp-tree@0.1.27", "", { "bin": { "regexp-tree": "bin/regexp-tree" } }, "sha512-iETxpjK6YoRWJG5o6hXLwvjYAoW+FEZn9os0PD/b6AP6xQwsa/Y7lCVgIixBbUPMfhu+i2LtdeAqVTgGlQarfA=="], diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 43f0267f6..488ff65e4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -45,7 +45,7 @@ - Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately - Fixed the prewalk plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls - Fixed launch tool rendering stacking a stale pending header over a bare `✓ Launch` line and raw text: the tool now uses a merged registry renderer with one per-op status header (op, target, `state · pid · uptime` meta), stripped log cursor suffixes, capped collapsed log/list previews, and a launch tool glyph -- Fixed `launch logs` flattening PTY control sequences into repeated debugger frames: raw daemon output is now replayed through the shared xterm screen renderer used by Bash PTY mode, preserving cursor updates, colors, and text styles while model-facing log text remains sanitized. +- Fixed `launch logs` flattening PTY control sequences into repeated or diagonally wrapped debugger fragments: the bounded raw PTY stream is now replayed before row selection through the shared xterm screen renderer used by Bash PTY mode, blank terminal cells retain their columns, and the final colored/styled viewport renders in the same bordered output block as Bash while model-facing text remains sanitized - Fixed confusing launch start/wait results when readiness timed out with the log pattern already matched (readiness needs log AND port): the result printed a contradictory `Ready: ` next to `Readiness timed out` without naming the failing condition. Daemon snapshots now carry the unmet conditions (`readyPending`), and start/wait results state exactly what never happened (e.g. `port 3100 on 127.0.0.1 never accepted connections`); the TUI shows a `waiting on port` badge on starting daemons - Fixed the in-process `stat` builtin mangling BSD-style invocations like `stat -f "%Sm %N" file` (macOS muscle memory): GNU `-f` means `--file-system`, so the format string was treated as a file operand — printing filesystem info for the real operands and erroring with `cannot read file system information for '%Sm %N'`. A `-f` whose format value contains `%` is now detected as BSD syntax and translated to the GNU equivalent (`%Sm`→`%y`, `%N`→`%n`, `%z`→`%s`, epoch/`S`-form times, owner/group/permission and `H`/`L` sub-field directives, `-L`/`-n`/`-q`/`-F` flag clusters, with `%n`/`%t` as literal newline/tab); directives with no GNU counterpart fail with a clear `unsupported BSD format directive` error - Fixed the remaining GNU-flavored shell builtins that broke under macOS/BSD muscle memory, using the same unambiguous-detection approach as the `stat` fix (only invocations that are invalid or nonsensical under GNU semantics are reinterpreted; unsupported BSD forms fail loudly instead of producing wrong output): `date -r ` formats the epoch when no such file exists (GNU `-r FILE` mtime preserved), signed `date -v±N` adjustments translate to `-d` relative dates and `-j` is accepted (`-j -f` strptime parse mode and field-set `-v` error clearly); `sed -i '' 's/…/…/' file` drops the BSD empty backup-suffix token instead of treating it as the script; `mktemp -t prefix` without X's creates `$TMPDIR/prefix.XXXXXXXXXX` (the GNU `too few X's` error path); `tail -r` reverses input by delegating to `tac` (with `-n`/`-c`/`-f` combinations erroring clearly); `find -E` maps to `-regextype posix-extended` ahead of the expression; `base64 -D` decodes as an alias of `-d`; and `ln -sfh` works via a `-h` alias of `--no-dereference` (clap's `-h` help short is dropped to match real GNU/BSD ln; `--help` unchanged) diff --git a/packages/harbor-manager/package.json b/packages/harbor-manager/package.json index 1220e4750..3b7e14fee 100644 --- a/packages/harbor-manager/package.json +++ b/packages/harbor-manager/package.json @@ -45,7 +45,8 @@ "@types/d3-scale": "^4.0.9", "@types/d3-shape": "^3.1.7", "@types/react": "^19.1.0", - "@types/react-dom": "^19.1.0" + "@types/react-dom": "^19.1.0", + "react-refresh": "^0.18.0" }, "engines": { "bun": ">=1.3.14" diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index 39ecb7447..9c8045d51 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -182,10 +182,9 @@ export class ManagerServer { // Bun bundles the dashboard (React + TSX) from the HTML import and // serves it on the same port as the API — one process, no Vite. routes: { "/": indexHtml }, - // Only `hmr`: Bun's `console: true` mirror opens a server-read stream - // over the dev client that a `--hot` reload's force-close tears down - // mid-read, surfacing an unhandled `AbortError: ERR_STREAM_RELEASE_LOCK` - // that crashes the process. + // Only `hmr`: the `console: true` mirror adds another dev-client + // stream with the same teardown hazard (see isDevStreamTeardown) + // for little value. development: process.env.NODE_ENV !== "production" && { hmr: true }, fetch: request => this.#route(request), }); @@ -628,22 +627,45 @@ function readTextTail(file: string, cap: number): string { } } +/** + * Bun's dev server (HMR websocket, browser error reports, console mirror) + * reads client streams that a tab disconnect or `--hot` reload tears down + * mid-read. The resulting `AbortError: ERR_STREAM_RELEASE_LOCK` surfaces as + * an unhandled rejection from Bun internals — fatal by default, which would + * kill the manager and orphan every running benchmark job. + */ +function isDevStreamTeardown(err: unknown): boolean { + return err instanceof Error && (err as Error & { code?: string }).code === "ERR_STREAM_RELEASE_LOCK"; +} + if (import.meta.main) { // `bun --hot` re-evaluates this module in-place: retire the previous // instance first, or its sync ticker and sqlite connection leak per reload. - const host = globalThis as typeof globalThis & { __harborManagerServer?: ManagerServer }; + const host = globalThis as typeof globalThis & { + __harborManagerServer?: ManagerServer; + __harborManagerHooks?: boolean; + }; await host.__harborManagerServer?.stop(); const { port, jobsDir } = parseServerArgs(process.argv.slice(2)); const manager = new ManagerServer(jobsDir); host.__harborManagerServer = manager; const server = manager.start(port); process.stdout.write(`harbor-manager listening on http://localhost:${server.port} (jobs: ${jobsDir})\n`); - const shutdown = async () => { - await manager.stop(); - process.exit(0); - }; - process.on("SIGINT", shutdown); - process.on("SIGTERM", shutdown); + // Process-wide hooks register once; `--hot` re-evals reuse them via `host`. + if (!host.__harborManagerHooks) { + host.__harborManagerHooks = true; + const shutdown = async () => { + await host.__harborManagerServer?.stop(); + process.exit(0); + }; + process.on("SIGINT", shutdown); + process.on("SIGTERM", shutdown); + process.on("unhandledRejection", err => { + if (isDevStreamTeardown(err)) { + process.stderr.write("ignored dev-server stream teardown (ERR_STREAM_RELEASE_LOCK)\n"); + return; + } + throw err; // preserve fail-fast for real bugs + }); + } } - -// hot-reload-probe From f9f6ed9e8d42174a4f370b63981c0a43c89f750e Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 23:26:33 +0200 Subject: [PATCH 22/28] feat(coding-agent): replaced legacy `pi/` role alias prefix with - Replaced legacy `pi/` role alias prefix with canonical `@` syntax across model resolution, documentation, and tests. - Added support for bare `*` default alias and multiple alias prefix detection with custom role resolution in `resolveConfiguredRolePattern()`. - Enhanced thinking suffix parsing to accept unambiguous abbreviations (minimum 2 characters) for effort and level selectors. - Extended `resolveCliModel()` and `filterAvailableModelsByEnabledPatterns()` to accept settings parameter for role alias resolution from `--model` flag. --- docs/extensions.md | 2 +- docs/local-models.md | 2 +- docs/models.md | 4 +- docs/settings.md | 2 +- docs/tools/eval.md | 6 +- docs/tools/inspect_image.md | 2 +- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/config/model-resolver.ts | 141 ++++++++++++++---- .../coding-agent/src/config/model-roles.ts | 14 ++ .../src/config/settings-schema.ts | 6 +- .../src/eval/__tests__/agent-bridge.test.ts | 4 +- .../eval/__tests__/completion-bridge.test.ts | 2 +- .../src/eval/completion-bridge.ts | 8 +- .../src/extensibility/extensions/model-api.ts | 2 +- .../src/extensibility/extensions/types.ts | 2 +- packages/coding-agent/src/main.ts | 10 +- .../src/prompts/agents/designer.md | 2 +- .../src/prompts/agents/librarian.md | 2 +- .../src/prompts/agents/reviewer.md | 2 +- .../coding-agent/src/prompts/agents/scout.md | 2 +- packages/coding-agent/src/sdk.ts | 2 +- .../coding-agent/src/session/agent-session.ts | 2 +- .../src/slash-commands/builtin-registry.ts | 2 +- packages/coding-agent/src/task/agents.ts | 4 +- packages/coding-agent/src/thinking.ts | 13 +- packages/coding-agent/src/tiny/models.ts | 14 +- .../coding-agent/src/tools/inspect-image.ts | 4 +- .../coding-agent/src/tts/speech-enhancer.ts | 4 +- .../src/utils/image-vision-fallback.ts | 6 +- packages/coding-agent/src/vibe/runtime.ts | 4 +- .../test/bundled-agent-parsing.test.ts | 4 +- ...mmit-model-selection-role-thinking.test.ts | 2 +- .../extensibility/ext-model-query.test.ts | 2 +- .../coding-agent/test/model-resolver.test.ts | 137 ++++++++++++++--- .../role-thinking-helper-propagation.test.ts | 4 +- .../test/sdk-model-selection.test.ts | 4 +- .../test/task/executor-pass-through.test.ts | 4 +- .../test/tools/inspect-image.test.ts | 2 +- 38 files changed, 313 insertions(+), 117 deletions(-) diff --git a/docs/extensions.md b/docs/extensions.md index 7702c1767..786f66284 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -167,7 +167,7 @@ Handlers and tool `execute` receive `ctx` with: - `list()` — authenticated models available this session. - `current()` — the live session model (read lazily, so it reflects `/model` switches). -- `resolve(spec)` — a model string (`provider/id`, bare id) or role alias (`pi/slow`, a configured role) → `Model`, honoring the same settings-backed aliases and match preferences as `--model`. Returns `undefined` when nothing matches. +- `resolve(spec)` — a model string (`provider/id`, bare id) or role alias (`@slow`, a configured role) → `Model`, honoring the same settings-backed aliases and match preferences as `--model`. Returns `undefined` when nothing matches. - `family(model)` — an opaque lineage token for "same family?" checks (Claude point releases share a token; Claude and GPT differ). Compare it; don't persist it (the vocabulary tracks new releases). ```ts diff --git a/docs/local-models.md b/docs/local-models.md index 2a5a494d9..41de745f1 100644 --- a/docs/local-models.md +++ b/docs/local-models.md @@ -75,7 +75,7 @@ they opt in. | flan-t5-small | Rejected — just echoes the input | **Shipped local options**: `lfm2-350m`, `qwen3-0.6b`, `gemma-270m`, `qwen2.5-0.5b`, `lfm2-700m`. -**Default**: `online` (pi/smol). +**Default**: `online` (@smol). ## Task 2: Mnemopi memory (`providers.memoryModel`) diff --git a/docs/models.md b/docs/models.md index b6d3cc70d..cd4663b4d 100644 --- a/docs/models.md +++ b/docs/models.md @@ -446,9 +446,9 @@ Supported model roles: - `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor` -The `tiny` role overrides the online model used for lightweight background tasks (session titles, memory, `auto`-thinking difficulty classification, unexpected-stop detection); when unset, these fall back to `pi/smol`. Pick one in `/models`. +The `tiny` role overrides the online model used for lightweight background tasks (session titles, memory, `auto`-thinking difficulty classification, unexpected-stop detection); when unset, these fall back to `@smol`. Pick one in `/models`. -Role aliases like `pi/smol` expand through `settings.modelRoles`. Each role value can also append a thinking selector such as `:minimal`, `:low`, `:medium`, or `:high`. +Role aliases like `@smol` expand through `settings.modelRoles`; `*` selects `@default`. Quote `@` aliases in YAML values (`fable: "@slow"`). Each role value can also append a thinking selector such as `:minimal`, `:low`, `:medium`, or `:high`. If a role points at another role, the target model still inherits normally and any explicit suffix on the referring role wins for that role-specific use. diff --git a/docs/settings.md b/docs/settings.md index e56b620de..76be155f6 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -308,7 +308,7 @@ enabledModels: | Key | Type | Default | Notes | |---|---|---|---| -| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor`. The `tiny` role overrides the online model for lightweight background tasks (titles, memory, auto-thinking, unexpected-stop), else `pi/smol`. Per-role env/flags exist only for `--model`/`--smol`/`--slow`/`--plan`; configure the advisor with `modelRoles.advisor`. | +| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `tiny`, `task`, `advisor`. The `tiny` role overrides the online model for lightweight background tasks (titles, memory, auto-thinking, unexpected-stop), else `@smol`. Per-role env/flags exist only for `--model`/`--smol`/`--slow`/`--plan`; configure the advisor with `modelRoles.advisor`. | | `modelTags` | record | `{}` | Custom role/tag metadata; can introduce additional roles. | | `modelProviderOrder` | array | `[]` | Preferred provider order when a model id is ambiguous. | | `cycleOrder` | array | `["smol","default","slow"]` | Roles cycled by the model switcher. | diff --git a/docs/tools/eval.md b/docs/tools/eval.md index be4dda38a..e41a97ee2 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -179,9 +179,9 @@ Both runtimes expose `completion()` — a single stateless completion against a - JS: `await completion(prompt, { model?, system?, schema? })` - Python: `completion(prompt, *, model="default", system=None, schema=None)` - `model` selects a tier (default `"default"`): - - `"smol"` → `pi/smol` role (fast / cheap) - - `"default"` → the session's active model, falling back to the `pi/default` role - - `"slow"` → `pi/slow` role; requests high reasoning effort only on reasoning-capable models + - `"smol"` → `@smol` role (fast / cheap) + - `"default"` → the session's active model, falling back to the `@default` role + - `"slow"` → `@slow` role; requests high reasoning effort only on reasoning-capable models - `system` (optional) supplies a system prompt. - `schema` (optional) is a plain JSON-Schema object. When present, the model is forced to call a single synthetic `respond` tool with that schema (loose, non-strict), and the helper returns the parsed object. When absent, the helper returns the completion string. - Errors surface as exceptions: unresolved tier, missing API key, an `error`/`aborted` stop reason, or empty output each raise. diff --git a/docs/tools/inspect_image.md b/docs/tools/inspect_image.md index aaea1fb08..36454eb70 100644 --- a/docs/tools/inspect_image.md +++ b/docs/tools/inspect_image.md @@ -40,7 +40,7 @@ TUI rendering adds presentation-only truncation from `packages/coding-agent/src/ ## Flow 1. `InspectImageTool.execute(...)` rejects immediately if `images.blockImages` is enabled in session settings. 2. It reads `session.modelRegistry`; missing registry, empty registry, missing API key, or unresolved model each raise `ToolError` from `packages/coding-agent/src/tools/inspect-image.ts`. -3. Model selection tries, in order, `pi/vision`, `pi/default`, the active model string from the session, then `availableModels[0]`. `expandRoleAlias(...)` and `resolveModelFromString(...)` handle each lookup. +3. Model selection tries, in order, `@vision`, `@default`, the active model string from the session, then `availableModels[0]`. `expandRoleAlias(...)` and `resolveModelFromString(...)` handle each lookup. 4. The chosen model must advertise `input.includes("image")`; otherwise execution fails before reading the file. 5. `loadImageInput(...)` in `packages/coding-agent/src/utils/image-loading.ts` resolves the path with `resolveReadPath(...)`, detects MIME type with `readImageMetadata(...)`, and rejects files larger than `MAX_IMAGE_INPUT_BYTES` (`20 * 1024 * 1024`, 20 MiB) using `ImageInputTooLargeError`. 6. `readImageMetadata(...)` in `packages/utils/src/mime.ts` inspects file headers only. Supported detected MIME types are `image/png`, `image/jpeg`, `image/gif`, and `image/webp`. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 488ff65e4..8175722cf 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -35,6 +35,7 @@ ### Fixed - Fixed expanded (ctrl+O) streaming edit previews duplicating the tool box in terminal scrollback: an unbounded live diff scrolled above the native-scrollback commit boundary mid-stream, freezing a stale preview snapshot that the finalized render then recommitted below. Expanded previews now use a viewport-sized tail window; the full diff still renders once the result finalizes. +- Fixed configured custom model roles resolving from `--model`; canonical role selectors now use `@role`, legacy selectors remain supported, and `*` selects the default role. Thinking suffixes split correctly off the bare `*` alias (`*:xhigh`), thinking selectors accept unambiguous abbreviations (`:xhi`, `:med`) on every surface, and role aliases now also work in `--models` and `enabledModels` scopes (each role contributes its resolved model). - Fixed quadratic growth of `--mode json` logs by eliding redundant message snapshots and payloads - Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache - Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 32ed6116f..08fdb5e18 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -37,7 +37,14 @@ import { resolveThinkingLevelForModel, } from "../thinking"; import { isAuthenticated, kNoAuth, type ModelRegistry } from "./model-registry"; -import { MODEL_ROLE_IDS, type ModelRole } from "./model-roles"; +import { + DEFAULT_MODEL_ROLE_ALIAS, + formatModelRoleAlias, + LEGACY_MODEL_ROLE_ALIAS_PREFIX, + MODEL_ROLE_ALIAS_PREFIX, + MODEL_ROLE_IDS, + type ModelRole, +} from "./model-roles"; import type { Settings } from "./settings"; function isKnownProvider(provider: string): provider is KnownProvider { @@ -109,8 +116,7 @@ function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): Co * level / `:auto` sentinel); `base` then has the suffix stripped. Otherwise * `base` is the input. * `minColonIndex` requires the colon to appear strictly after that index — - * role-alias callers pass `PREFIX_MODEL_ROLE.length` so the base is at least - * as long as the `pi/` prefix. + * role-alias callers pass the matched alias prefix length. */ function splitThinkingSuffix( pattern: string, @@ -850,17 +856,32 @@ export function parseModelPattern( ); } -const PREFIX_MODEL_ROLE = "pi/"; const DEFAULT_MODEL_ROLE = "default"; +const MODEL_ROLE_ALIAS_PREFIXES = [MODEL_ROLE_ALIAS_PREFIX, LEGACY_MODEL_ROLE_ALIAS_PREFIX]; -function getModelRoleAlias(value: string): ModelRole | undefined { +function isModelRole(role: string): role is ModelRole { + return (MODEL_ROLE_IDS as string[]).includes(role); +} + +/** + * Minimum colon index for splitting a `:` suffix off a role alias, or + * `undefined` when `value` is not role-alias shaped. Doubles as the slice + * offset of the role name for prefixed aliases (`@role`, `pi/role`); the bare + * `*` default alias returns 0 because its colon sits immediately after the + * one-character token (`*:xhigh`). + */ +function modelRoleAliasPrefixLength(value: string): number | undefined { + if (value === DEFAULT_MODEL_ROLE_ALIAS || value.startsWith(`${DEFAULT_MODEL_ROLE_ALIAS}:`)) return 0; + return MODEL_ROLE_ALIAS_PREFIXES.find(prefix => value.startsWith(prefix))?.length; +} + +function getModelRoleAlias(value: string, settings?: Settings): string | undefined { const normalized = value.trim(); - if (!normalized.startsWith(PREFIX_MODEL_ROLE)) return undefined; + const prefixLength = modelRoleAliasPrefixLength(normalized); + if (prefixLength === undefined) return undefined; - const candidate = normalized.slice(PREFIX_MODEL_ROLE.length); - for (const role of MODEL_ROLE_IDS) { - if (candidate === role) return role; - } + const candidate = normalized === DEFAULT_MODEL_ROLE_ALIAS ? DEFAULT_MODEL_ROLE : normalized.slice(prefixLength); + if (isModelRole(candidate) || settings?.getModelRole(candidate) !== undefined) return candidate; return undefined; } @@ -871,7 +892,14 @@ function normalizeModelPatternList(value: string | string[] | undefined): string } function isSessionInheritedAgentPattern(value: string): boolean { - return value === DEFAULT_MODEL_ROLE || value === `${PREFIX_MODEL_ROLE}${DEFAULT_MODEL_ROLE}` || value === "pi/task"; + return ( + value === DEFAULT_MODEL_ROLE || + value === formatModelRoleAlias(DEFAULT_MODEL_ROLE) || + value === DEFAULT_MODEL_ROLE_ALIAS || + value === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}${DEFAULT_MODEL_ROLE}` || + value === formatModelRoleAlias("task") || + value === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}task` + ); } function shouldInheritDefaultBeforePriority(role: ModelRole): boolean { @@ -904,7 +932,7 @@ function resolveDefaultInheritedPatterns( configuredDefault: string | undefined, roleDefaults: string[], settings: Settings | undefined, - visited: Set, + visited: Set, ): string[] { if (!shouldInheritDefaultBeforePriority(role) || !configuredDefault) return []; @@ -912,13 +940,12 @@ function resolveDefaultInheritedPatterns( for (const pattern of normalizeModelPatternList(configuredDefault)) { const { base: aliasCandidate, level: thinkingLevel } = splitThinkingSuffix( pattern, - PREFIX_MODEL_ROLE.length, + modelRoleAliasPrefixLength(pattern) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length, MAX_THINKING_SUFFIX_OPTIONS, ); - const aliasRole = getModelRoleAlias(aliasCandidate); + const aliasRole = getModelRoleAlias(aliasCandidate, settings); if (aliasRole === role) { - // Self-alias (e.g. modelRoles.default = "pi/smol") would loop back to the - // same unset role; collapse straight to the built-in priority chain. + // Self-alias (e.g. modelRoles.default = "@smol") would loop back to the resolved.push( ...(thinkingLevel ? roleDefaults.map(defaultPattern => `${defaultPattern}:${thinkingLevel}`) @@ -927,8 +954,7 @@ function resolveDefaultInheritedPatterns( continue; } if (aliasRole && !visited.has(aliasRole)) { - // Cross-role alias (e.g. modelRoles.default = "pi/slow"): resolve the - // target role's patterns now so downstream one-layer expanders see + // Cross-role alias (e.g. modelRoles.default = "@slow"): resolve the // concrete model patterns instead of another role alias. const recursed = resolveConfiguredRolePattern(pattern, settings, new Set(visited)); if (recursed && recursed.length > 0) { @@ -944,27 +970,29 @@ function resolveDefaultInheritedPatterns( function resolveConfiguredRolePattern( value: string, settings?: Settings, - visited: Set = new Set(), + visited: Set = new Set(), ): string[] | undefined { const normalized = value.trim(); if (!normalized) return undefined; const { base: aliasCandidate, level: thinkingLevel } = splitThinkingSuffix( normalized, - PREFIX_MODEL_ROLE.length, + modelRoleAliasPrefixLength(normalized) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length, MAX_THINKING_SUFFIX_OPTIONS, ); - const role = getModelRoleAlias(aliasCandidate); + const role = getModelRoleAlias(aliasCandidate, settings); if (!role) return [normalized]; if (visited.has(role)) return undefined; visited.add(role); const configured = settings?.getModelRole(role)?.trim(); const configuredDefault = settings?.getModelRole(DEFAULT_MODEL_ROLE)?.trim(); - const roleDefaults = rolePriorityDefaults(role); + const roleDefaults = isModelRole(role) ? rolePriorityDefaults(role) : []; const resolved = configured ? normalizeModelPatternList(configured) - : resolveDefaultInheritedPatterns(role, configuredDefault, roleDefaults, settings, visited); + : isModelRole(role) + ? resolveDefaultInheritedPatterns(role, configuredDefault, roleDefaults, settings, visited) + : roleDefaults; if (resolved.length === 0) { resolved.push(...roleDefaults); } @@ -976,7 +1004,7 @@ function resolveConfiguredRolePattern( } /** - * Expand a role alias like "pi/smol" to the configured model string. + * Expand a role alias like "@smol" to the configured model string. */ export function expandRoleAlias(value: string, settings?: Settings): string { const normalized = value.trim(); @@ -1014,8 +1042,13 @@ export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOp const singleAgentPattern = normalizedAgentPatterns.length === 1 ? normalizedAgentPatterns[0] : undefined; const agentInheritsSessionModel = singleAgentPattern ? isSessionInheritedAgentPattern(singleAgentPattern) : false; if (configuredAgentPatterns.length > 0) { + if ( + singleAgentPattern === formatModelRoleAlias("task") || + singleAgentPattern === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}task` + ) { + return configuredAgentPatterns; + } if (!agentInheritsSessionModel) return configuredAgentPatterns; - if (singleAgentPattern === "pi/task") return configuredAgentPatterns; } const fallback = @@ -1102,12 +1135,16 @@ export function extractExplicitThinkingSelector( let current = normalized; while (!visited.has(current)) { visited.add(current); - const strictSelector = splitThinkingSuffix(current, PREFIX_MODEL_ROLE.length).level; + const rolePrefixLength = modelRoleAliasPrefixLength(current) ?? LEGACY_MODEL_ROLE_ALIAS_PREFIX.length; + const strictSelector = splitThinkingSuffix(current, rolePrefixLength).level; if (strictSelector) { return strictSelector; } - const maxSelector = splitThinkingSuffix(current, PREFIX_MODEL_ROLE.length, MAX_THINKING_SUFFIX_OPTIONS).level; - if (maxSelector && (current.startsWith(PREFIX_MODEL_ROLE) || !isLiteralModelSelector(current, options))) { + const maxSelector = splitThinkingSuffix(current, rolePrefixLength, MAX_THINKING_SUFFIX_OPTIONS).level; + if ( + maxSelector && + (modelRoleAliasPrefixLength(current) !== undefined || !isLiteralModelSelector(current, options)) + ) { return maxSelector; } const expanded = expandRoleAlias(current, settings).trim(); @@ -1278,7 +1315,7 @@ export function resolveAdvisorRoleSelection( settings: Settings, availableModels: Model[], ): { model: Model; thinkingLevel?: ConfiguredThinkingLevel } | undefined { - const resolved = resolveModelRoleValue(`${PREFIX_MODEL_ROLE}advisor`, availableModels, { + const resolved = resolveModelRoleValue(formatModelRoleAlias("advisor"), availableModels, { settings, matchPreferences: getModelMatchPreferences(settings), }); @@ -1300,6 +1337,7 @@ export async function resolveModelScope( patterns: string[], modelRegistry: Pick, preferences?: ModelMatchPreferences, + settings?: Settings, ): Promise { const availableModels = modelRegistry.getAvailable(); const context = buildPreferenceContext(availableModels, preferences); @@ -1337,6 +1375,25 @@ export async function resolveModelScope( continue; } + // Role aliases (`@smol`, `pi/slow`) resolve to the role's single concrete + // model — not its whole fallback chain — so a role contributes one scope + // entry exactly like `--model` would pick. (Bare `*` stays a match-all + // glob above; scope semantics, not the default-role alias.) + if (settings && modelRoleAliasPrefixLength(pattern) !== undefined) { + const resolved = resolveModelRoleValue(pattern, availableModels, { settings, matchPreferences: preferences }); + if (resolved.warning) logger.warn(resolved.warning); + if (!resolved.model) { + logger.warn(`No models match pattern "${pattern}"`); + continue; + } + if (resolved.thinkingLevel === AUTO_THINKING) { + addScopedModel(resolved.model, undefined, false); + } else { + addScopedModel(resolved.model, resolved.thinkingLevel, resolved.explicitThinkingLevel); + } + continue; + } + const { model, thinkingLevel, warning, explicitThinkingLevel } = parseModelPatternWithContext( pattern, availableModels, @@ -1386,7 +1443,7 @@ export async function resolveAllowedModels( if (!patterns || patterns.length === 0) { return available; } - const scoped = await resolveModelScope(patterns, modelRegistry, preferences); + const scoped = await resolveModelScope(patterns, modelRegistry, preferences, settings); if (scoped.length === 0) { return []; } @@ -1415,6 +1472,7 @@ export async function resolveAllowedModels( export function filterAvailableModelsByEnabledPatterns( available: Model[], patterns: readonly string[], + settings?: Settings, ): Model[] { if (patterns.length === 0) return available; @@ -1432,6 +1490,13 @@ export function filterAvailableModelsByEnabledPatterns( continue; } + // Mirror resolveModelScope: role aliases resolve to the role's model. + if (settings && modelRoleAliasPrefixLength(pattern) !== undefined) { + const { model } = resolveModelRoleValue(pattern, available, { settings }); + if (model) addAllowed(model); + continue; + } + const { model } = parseModelPatternWithContext(pattern, available, context); if (model) { addAllowed(model); @@ -1456,9 +1521,10 @@ export function resolveCliModel(options: { cliProvider?: string; cliModel?: string; modelRegistry: CliModelRegistry; + settings?: Settings; preferences?: ModelMatchPreferences; }): ResolveCliModelResult { - const { cliProvider, cliModel, modelRegistry, preferences } = options; + const { cliProvider, cliModel, modelRegistry, settings, preferences } = options; if (!cliModel) { return { model: undefined, selector: undefined, warning: undefined, error: undefined }; @@ -1474,6 +1540,19 @@ export function resolveCliModel(options: { }; } + if (!cliProvider && modelRoleAliasPrefixLength(cliModel) !== undefined) { + const resolved = resolveModelRoleValue(cliModel, availableModels, { settings, matchPreferences: preferences }); + if (resolved.model) { + return { + model: resolved.model, + selector: formatModelString(resolved.model), + thinkingLevel: resolved.thinkingLevel, + warning: resolved.warning, + error: undefined, + }; + } + } + const providerMap = new Map(); for (const model of availableModels) { providerMap.set(model.provider.toLowerCase(), model.provider); diff --git a/packages/coding-agent/src/config/model-roles.ts b/packages/coding-agent/src/config/model-roles.ts index 3ab5e9489..6d97d94b4 100644 --- a/packages/coding-agent/src/config/model-roles.ts +++ b/packages/coding-agent/src/config/model-roles.ts @@ -5,6 +5,20 @@ import { isValidThemeColor, type ThemeColor } from "../modes/theme/theme"; import type { Settings } from "./settings"; +/** Canonical prefix for a configured model role selector. */ +export const MODEL_ROLE_ALIAS_PREFIX = "@"; + +/** Legacy prefix accepted for backwards-compatible role selectors. */ +export const LEGACY_MODEL_ROLE_ALIAS_PREFIX = "pi/"; + +/** Shorthand selector for the default model role. */ +export const DEFAULT_MODEL_ROLE_ALIAS = "*"; + +/** Format a model role as its canonical selector. */ +export function formatModelRoleAlias(role: string): string { + return `${MODEL_ROLE_ALIAS_PREFIX}${role}`; +} + export type ModelRole = | "default" | "smol" diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 468780b61..aa08b363d 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -2639,14 +2639,14 @@ export const SETTINGS_SCHEMA = { group: "Mnemopi", label: "Mnemopi LLM Mode", description: - "Use no LLM, the online tiny model (the TINY role from /models, else pi/smol), or a remote OpenAI-compatible endpoint", + "Use no LLM, the online tiny model (the TINY role from /models, else @smol), or a remote OpenAI-compatible endpoint", condition: "mnemopiActive", options: [ { value: "none", label: "None", description: "Disable Mnemopi LLM-backed extraction" }, { value: "smol", label: "Online (tiny)", - description: "Use the online tiny model (the TINY role from /models, else pi/smol)", + description: "Use the online tiny model (the TINY role from /models, else @smol)", }, { value: "remote", label: "Remote", description: "Use the Mnemopi remote LLM settings below" }, ], @@ -4600,7 +4600,7 @@ export const SETTINGS_SCHEMA = { group: "Tiny Model", label: "Tiny Model", description: - "Session-title model: online (the TINY role from /models, else pi/smol) by default, or a local on-device model", + "Session-title model: online (the TINY role from /models, else @smol) by default, or a local on-device model", options: TINY_TITLE_MODEL_OPTIONS, }, }, diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 320c84fb2..bff2e44d1 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -28,7 +28,7 @@ const taskAgent = { systemPrompt: "Run the task.", source: "bundled", spawns: "*", - model: ["pi/task"], + model: ["@task"], } satisfies AgentDefinition; const reviewerAgent = { @@ -36,7 +36,7 @@ const reviewerAgent = { description: "Reviewer agent", systemPrompt: "Review the task.", source: "bundled", - model: ["pi/smol"], + model: ["@smol"], } satisfies AgentDefinition; interface SessionOptions { diff --git a/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts index 0b4422021..59ad6c844 100644 --- a/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts @@ -193,7 +193,7 @@ describe("runEvalCompletion", () => { expect(resolved).toEqual(["p/smol", "p/default", "p/slow"]); }); - it("prefers the session active model for the default tier, falling back to pi/default", async () => { + it("prefers the session active model for the default tier, falling back to @default", async () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); const session = makeSession({ available: [SMOL, DEFAULT, SLOW], activeModel: "p/slow" }); diff --git a/packages/coding-agent/src/eval/completion-bridge.ts b/packages/coding-agent/src/eval/completion-bridge.ts index 52f04f254..03bf6d778 100644 --- a/packages/coding-agent/src/eval/completion-bridge.ts +++ b/packages/coding-agent/src/eval/completion-bridge.ts @@ -37,9 +37,9 @@ const STRUCTURED_TOOL_NAME = "respond"; type CompletionTier = "smol" | "default" | "slow"; const TIER_TO_PATTERN: Record = { - smol: "pi/smol", - default: "pi/default", - slow: "pi/slow", + smol: "@smol", + default: "@default", + slow: "@slow", }; const completionArgsSchema = type({ @@ -62,7 +62,7 @@ export interface EvalCompletionResult { /** * Resolve a tier to a concrete {@link Model}. `default` prefers the session's - * active model and falls back to the `pi/default` role; `smol`/`slow` resolve + * active model and falls back to the `@default` role; `smol`/`slow` resolve * their respective role patterns. Returns `undefined` when nothing matches. */ function resolveTierModel(tier: CompletionTier, session: ToolSession): Model | undefined { diff --git a/packages/coding-agent/src/extensibility/extensions/model-api.ts b/packages/coding-agent/src/extensibility/extensions/model-api.ts index 3d9560c7e..6a6aeb075 100644 --- a/packages/coding-agent/src/extensibility/extensions/model-api.ts +++ b/packages/coding-agent/src/extensibility/extensions/model-api.ts @@ -25,7 +25,7 @@ export function createExtensionModelQuery( return { list: () => modelRegistry.getAvailable(), current: () => getModel(), - // resolveModelRoleValue expands a role alias (`pi/slow`) to its full configured + // resolveModelRoleValue expands a role alias (`@slow`) to its full configured // priority list and tries each pattern — the same path core selection uses — so a // fallback model lower in the list still resolves. Plain model strings pass through // as a single pattern. diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index ee079659b..97a1fccbf 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -391,7 +391,7 @@ export interface ExtensionModelQuery { /** The current session model, if one is set. */ current(): Model | undefined; /** - * Resolve a model string (`provider/id`, bare id) or role alias (`pi/slow`, a + * Resolve a model string (`provider/id`, bare id) or role alias (`@slow`, a * configured role) to a Model, using the same settings-backed aliases and match * preferences as core selection. Thinking/routing suffixes are accepted and resolved * to the base model (pass effort separately). Returns undefined when nothing matches. diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 41943669f..9fa982b20 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -852,6 +852,7 @@ export async function buildSessionOptions( cliProvider: parsed.provider, cliModel: parsed.model, modelRegistry, + settings: activeSettings, preferences: modelMatchPreferences, }); if (resolved.warning) { @@ -914,13 +915,13 @@ export async function buildSessionOptions( ? true : activeSettings.get("prewalk.enabled"); if (prewalkEnabled) { - const rolePattern = expandRoleAlias(parsed.prewalkInto ?? "pi/smol", activeSettings); + const rolePattern = expandRoleAlias(parsed.prewalkInto ?? "@smol", activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? "pi/smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? "@smol"}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); @@ -932,13 +933,13 @@ export async function buildSessionOptions( throw new Error("--plan-yolo-into requires --plan-yolo"); } if (parsed.planYolo) { - const rolePattern = expandRoleAlias(parsed.planYoloInto ?? "pi/smol", activeSettings); + const rolePattern = expandRoleAlias(parsed.planYoloInto ?? "@smol", activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.planYoloInto ?? "pi/smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.planYoloInto ?? "@smol"}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); @@ -1183,6 +1184,7 @@ export async function runRootCommand( modelPatterns, modelRegistry, modelMatchPreferences, + settingsInstance, ); } diff --git a/packages/coding-agent/src/prompts/agents/designer.md b/packages/coding-agent/src/prompts/agents/designer.md index 72091b563..f45910b69 100644 --- a/packages/coding-agent/src/prompts/agents/designer.md +++ b/packages/coding-agent/src/prompts/agents/designer.md @@ -1,7 +1,7 @@ --- name: designer description: UI/UX specialist for design implementation, review, visual refinement -model: pi/designer +model: "@designer" --- Implement and review UI designs. Edit files, create components, run commands when needed. diff --git a/packages/coding-agent/src/prompts/agents/librarian.md b/packages/coding-agent/src/prompts/agents/librarian.md index d3bb1c40a..dc7764013 100644 --- a/packages/coding-agent/src/prompts/agents/librarian.md +++ b/packages/coding-agent/src/prompts/agents/librarian.md @@ -2,7 +2,7 @@ name: librarian description: Researches external libraries and APIs by reading source code. Returns definitive, source-verified answers. tools: read, grep, glob, bash, lsp, web_search, ast_grep -model: pi/smol +model: "@smol" thinking-level: minimal read-summarize: false output: diff --git a/packages/coding-agent/src/prompts/agents/reviewer.md b/packages/coding-agent/src/prompts/agents/reviewer.md index edb4e51a2..11a468ddf 100644 --- a/packages/coding-agent/src/prompts/agents/reviewer.md +++ b/packages/coding-agent/src/prompts/agents/reviewer.md @@ -3,7 +3,7 @@ name: reviewer description: "Code review specialist for quality/security analysis" tools: read, grep, glob, bash, lsp, web_search, ast_grep spawns: scout -model: pi/slow +model: "@slow" output: properties: overall_correctness: diff --git a/packages/coding-agent/src/prompts/agents/scout.md b/packages/coding-agent/src/prompts/agents/scout.md index 277539b44..97af9b018 100644 --- a/packages/coding-agent/src/prompts/agents/scout.md +++ b/packages/coding-agent/src/prompts/agents/scout.md @@ -2,7 +2,7 @@ name: scout description: MUST be used for exploratory codebase research, rapid code analysis, and broad pattern searches. Fast read-only scout returning compressed context for handoff. tools: read, grep, glob, web_search -model: pi/smol +model: "@smol" thinking-level: medium read-summarize: false output: diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index cd7837a15..24c2ca15a 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1987,7 +1987,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } // Resolve deferred --model/subagent patterns now that extension models are - // registered. Expand role aliases (`pi/smol`) and comma chains to concrete + // registered. Expand role aliases (`@smol`) and comma chains to concrete // selectors first so deferred resolution accepts everything the immediate // path (resolveModelOverride → resolveModelRoleValue) accepts. if (!model && deferredModelPatterns.length > 0) { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 9065f723f..c6bc8b20b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -9649,7 +9649,7 @@ export class AgentSession { const all = this.#modelRegistry.getAvailable(); const patterns = this.settings.get("enabledModels"); if (!patterns || patterns.length === 0) return all; - return filterAvailableModelsByEnabledPatterns(all, patterns); + return filterAvailableModelsByEnabledPatterns(all, patterns, this.settings); } // ========================================================================= diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index fed8c2ce4..14843a6ee 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -466,7 +466,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ description: "Switch to a fast/cheap model at the next action (works even without --prewalk)", acpDescription: "Prewalk at the next action", handle: async (_command, runtime) => { - const rolePattern = expandRoleAlias("pi/smol", runtime.settings); + const rolePattern = expandRoleAlias("@smol", runtime.settings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry: runtime.session.modelRegistry, diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index caf8f46ad..2b071e4a4 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -50,7 +50,7 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ name: "task", description: "General-purpose subagent with full capabilities for delegated multi-step tasks", spawns: "*", - model: "pi/task", + model: "@task", thinkingLevel: AUTO_THINKING, }, template: taskMd, @@ -60,7 +60,7 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ frontmatter: { name: "sonic", description: "Low-reasoning agent for strictly mechanical updates or data collection only", - model: "pi/smol", + model: "@smol", thinkingLevel: Effort.Medium, }, template: taskMd, diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 0bf10760c..a872b93da 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -62,18 +62,25 @@ const THINKING_LEVEL_BY_SELECTOR: Readonly> = { }; function getOwnSelector(selectors: Readonly>, value: string | null | undefined): T | undefined { - return value === undefined || value === null || !Object.hasOwn(selectors, value) ? undefined : selectors[value]; + if (value === undefined || value === null) return undefined; + if (Object.hasOwn(selectors, value)) return selectors[value]; + // Accept unambiguous abbreviations (`xhi` → xhigh, `med` → medium) so every + // selector surface (`--thinking`, `:suffix`, role values) parses alike. + // Two-character minimum keeps single letters (`m`) from guessing. + if (value.length < 2) return undefined; + const matches = Object.keys(selectors).filter(selector => selector.startsWith(value)); + return matches.length === 1 ? selectors[matches[0]] : undefined; } /** - * Parses a provider-facing effort value. + * Parses a provider-facing effort value. Accepts unambiguous abbreviations. */ export function parseEffort(value: string | null | undefined): Effort | undefined { return getOwnSelector(EFFORT_BY_SELECTOR, value); } /** - * Parses an agent-local thinking selector. + * Parses an agent-local thinking selector. Accepts unambiguous abbreviations. */ export function parseThinkingLevel(value: string | null | undefined): ThinkingLevel | undefined { return getOwnSelector(THINKING_LEVEL_BY_SELECTOR, value); diff --git a/packages/coding-agent/src/tiny/models.ts b/packages/coding-agent/src/tiny/models.ts index 77c13426e..7500e2e25 100644 --- a/packages/coding-agent/src/tiny/models.ts +++ b/packages/coding-agent/src/tiny/models.ts @@ -1,4 +1,4 @@ -/** Default session-title model: the online pi/smol path (no local download / on-device inference). */ +/** Default session-title model: the online @smol path (no local download / on-device inference). */ export const ONLINE_TINY_TITLE_MODEL_KEY = "online"; /** Local model the `tiny-models` CLI downloads when none is named. Not the session-title default — that is {@link ONLINE_TINY_TITLE_MODEL_KEY}. */ export const DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY = "lfm2-700m"; @@ -87,9 +87,9 @@ void TINY_TITLE_MODEL_VALUES_MATCH_REGISTRY; export const TINY_TITLE_MODEL_OPTIONS = [ { value: ONLINE_TINY_TITLE_MODEL_KEY, - label: "Online (TINY role, else pi/smol)", + label: "Online (TINY role, else @smol)", description: - "Online title generation: the TINY model role (set one in /models) when assigned, otherwise the online fallback (commit role, then pi/smol). No local download or on-device inference.", + "Online title generation: the TINY model role (set one in /models) when assigned, otherwise the online fallback (commit role, then @smol). No local download or on-device inference.", }, ...TINY_TITLE_LOCAL_MODELS.map(model => ({ value: model.key, @@ -194,9 +194,9 @@ void TINY_MEMORY_MODEL_VALUES_MATCH_REGISTRY; export const TINY_MEMORY_MODEL_OPTIONS = [ { value: ONLINE_MEMORY_MODEL_KEY, - label: "Online (TINY role, else smol)", + label: "Online (TINY role, else @smol)", description: - "Use the online model: the TINY role from /models when set, otherwise pi/smol. No local model download or on-device inference.", + "Use the online model: the TINY role from /models when set, otherwise @smol. No local model download or on-device inference.", }, ...TINY_MEMORY_LOCAL_MODELS.map(model => ({ value: model.key, @@ -256,9 +256,9 @@ export type AutoThinkingModelKey = TinyMemoryModelKey; export const AUTO_THINKING_MODEL_OPTIONS = [ { value: ONLINE_AUTO_THINKING_MODEL_KEY, - label: "Online (TINY role, else smol)", + label: "Online (TINY role, else @smol)", description: - "Classify prompt difficulty online with the TINY role model (set one in /models) or pi/smol; no local download or on-device inference.", + "Classify prompt difficulty online with the TINY role model (set one in /models) or @smol; no local download or on-device inference.", }, ...TINY_MEMORY_LOCAL_MODELS.map(model => ({ value: model.key, diff --git a/packages/coding-agent/src/tools/inspect-image.ts b/packages/coding-agent/src/tools/inspect-image.ts index a270c5dbc..f090225f1 100644 --- a/packages/coding-agent/src/tools/inspect-image.ts +++ b/packages/coding-agent/src/tools/inspect-image.ts @@ -157,8 +157,8 @@ export class InspectImageTool implements AgentTool { try { const { settings, registry, sessionId } = this.#deps; - // `pi/tiny` expands a configured `modelRoles.tiny` and otherwise falls + // `@tiny` expands a configured `modelRoles.tiny` and otherwise falls // through tiny's alias to the smol priority chain — unlike bare role // lookup, this resolves even with no roles configured. - const model = resolveModelRoleValue("pi/tiny", registry.getAvailable(), { + const model = resolveModelRoleValue("@tiny", registry.getAvailable(), { settings, matchPreferences: getModelMatchPreferences(settings), }).model; diff --git a/packages/coding-agent/src/utils/image-vision-fallback.ts b/packages/coding-agent/src/utils/image-vision-fallback.ts index bf89a903d..26d34f032 100644 --- a/packages/coding-agent/src/utils/image-vision-fallback.ts +++ b/packages/coding-agent/src/utils/image-vision-fallback.ts @@ -97,7 +97,7 @@ function formatImageBlock(localUrl: string, description: string): string { /** * Resolve a vision-capable model, mirroring the inspect_image priority - * (`pi/vision` → `pi/default` → active → first image-capable available), but + * (`@vision` → `@default` → active → first image-capable available), but * never returning a text-only model. */ function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model | undefined { @@ -111,8 +111,8 @@ function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model | unde return model?.input.includes("image") ? model : undefined; }; return ( - resolvePattern("pi/vision") ?? - resolvePattern("pi/default") ?? + resolvePattern("@vision") ?? + resolvePattern("@default") ?? resolvePattern(deps.activeModelString) ?? available.find(model => model.input.includes("image")) ); diff --git a/packages/coding-agent/src/vibe/runtime.ts b/packages/coding-agent/src/vibe/runtime.ts index d7e1215aa..7d55b48c5 100644 --- a/packages/coding-agent/src/vibe/runtime.ts +++ b/packages/coding-agent/src/vibe/runtime.ts @@ -39,8 +39,8 @@ export type VibeCli = "fast" | "good"; /** * CLI flavor → bundled agent type. This IS the model-tier mapping: `sonic` - * carries `model: "pi/smol"` (the configured fast/low-latency role) and `task` - * carries `model: "pi/task"` (inherits the session's strong model). + * carries `model: "@smol"` (the configured fast/low-latency role) and `task` + * carries `model: "@task"` (inherits the session's strong model). * Resolution goes through {@link resolveAgentModelPatterns} exactly like a * `task` spawn, so `task.agentModelOverrides` and model-role settings apply. */ diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts index 0caf26767..8d206ca74 100644 --- a/packages/coding-agent/test/bundled-agent-parsing.test.ts +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -12,7 +12,7 @@ describe("bundled agent parsing", () => { expect(reviewer).toBeDefined(); expect(reviewer?.source).toBe("bundled"); - expect(reviewer?.model).toEqual(["pi/slow"]); + expect(reviewer?.model).toEqual(["@slow"]); expect(reviewer?.thinkingLevel).toBeUndefined(); }); @@ -20,7 +20,7 @@ describe("bundled agent parsing", () => { const task = getBundledAgent("task"); expect(task).toBeDefined(); - expect(task?.model).toEqual(["pi/task"]); + expect(task?.model).toEqual(["@task"]); expect(task?.thinkingLevel).toBe(AUTO_THINKING); }); diff --git a/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts b/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts index 4c315794c..14ae7530c 100644 --- a/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts +++ b/packages/coding-agent/test/commit-model-selection-role-thinking.test.ts @@ -34,7 +34,7 @@ describe("commit role thinking selection", () => { const settings = createSettings({ default: `${defaultModel.provider}/${defaultModel.id}:high`, commit: `${commitModel.provider}/${commitModel.id}:low`, - smol: "pi/default:minimal", + smol: "@default:minimal", }); const registry = { getAvailable: () => [defaultModel, commitModel], diff --git a/packages/coding-agent/test/extensibility/ext-model-query.test.ts b/packages/coding-agent/test/extensibility/ext-model-query.test.ts index db119ed73..c1d7c3469 100644 --- a/packages/coding-agent/test/extensibility/ext-model-query.test.ts +++ b/packages/coding-agent/test/extensibility/ext-model-query.test.ts @@ -60,7 +60,7 @@ describe("createExtensionModelQuery", () => { getModelRole: (role: string) => (role === "slow" ? "anthropic/claude-opus-4-8" : undefined), } as unknown as Settings; const q = createExtensionModelQuery(registry(), settings, () => undefined); - expect(q.resolve("pi/slow")).toBe(claude); + expect(q.resolve("@slow")).toBe(claude); }); test("family() groups a vendor's point releases and separates vendors", () => { diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index 4e2e5d4fc..5b3152e40 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -17,6 +17,7 @@ import { resolveModelRoleValue, resolveModelScope, } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { DEFAULT_MODEL_ROLE_ALIAS, LEGACY_MODEL_ROLE_ALIAS_PREFIX } from "@oh-my-pi/pi-coding-agent/config/model-roles"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; // Mock models for testing @@ -623,12 +624,12 @@ describe("parseModelPattern", () => { }); describe("resolveModelRoleValue", () => { - test("resolves pi/: by expanding role alias before parsing thinking", () => { + test("resolves @role: by expanding role alias before parsing thinking", () => { const settings = { getModelRole: (role: string) => (role === "smol" ? "openrouter/qwen/qwen3-coder:exacto" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/smol:high", allModels, { settings }); + const result = resolveModelRoleValue("@smol:high", allModels, { settings }); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("qwen/qwen3-coder:exacto"); @@ -636,12 +637,12 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); - test("resolves pi/:max by expanding role alias before parsing thinking", () => { + test("resolves @role:max by expanding role alias before parsing thinking", () => { const settings = { getModelRole: (role: string) => (role === "smol" ? "openai-codex/gpt-5.3-codex" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/smol:max", allModels, { settings }); + const result = resolveModelRoleValue("@smol:max", allModels, { settings }); expect(result.model?.provider).toBe("openai-codex"); expect(result.model?.id).toBe("gpt-5.3-codex"); @@ -650,12 +651,12 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); - test("resolves pi/default through configured default role alias", () => { + test("resolves @default through configured default role alias", () => { const settings = { getModelRole: (role: string) => (role === "default" ? "openrouter/qwen/qwen3-coder:exacto" : undefined), } as NonNullable[2]>["settings"]; - const result = resolveModelRoleValue("pi/default", allModels, { settings }); + const result = resolveModelRoleValue("@default", allModels, { settings }); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("qwen/qwen3-coder:exacto"); @@ -736,13 +737,13 @@ describe("resolveModelRoleValue", () => { }); }); describe("resolveAgentModelPatterns", () => { - test("falls back to the active session model when pi/task is unset", () => { + test("falls back to the active session model when @task is unset", () => { const settings = Settings.isolated({ modelRoles: { default: "anthropic/claude-sonnet-4-5" }, }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, activeModelPattern: "openai/gpt-4o", }); @@ -759,7 +760,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, activeModelPattern: "openai/gpt-4o", }); @@ -775,7 +776,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/task", + agentModel: "@task", settings, }); @@ -787,17 +788,17 @@ describe("resolveAgentModelPatterns", () => { modelRoles: { default: "local/llama" }, }); - expect(resolveAgentModelPatterns({ agentModel: "pi/smol", settings })).toEqual(["local/llama"]); - expect(resolveAgentModelPatterns({ agentModel: "pi/slow", settings })).toEqual(["local/llama"]); - expect(resolveAgentModelPatterns({ agentModel: "pi/designer", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@smol", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@slow", settings })).toEqual(["local/llama"]); + expect(resolveAgentModelPatterns({ agentModel: "@designer", settings })).toEqual(["local/llama"]); }); test("expands cross-role default aliases when inheriting for an unset role", () => { const settings = Settings.isolated({ - modelRoles: { default: "pi/slow", slow: "anthropic/claude-sonnet-4-5" }, + modelRoles: { default: "@slow", slow: "anthropic/claude-sonnet-4-5" }, }); - expect(resolveAgentModelPatterns({ agentModel: "pi/smol", settings })).toEqual(["anthropic/claude-sonnet-4-5"]); + expect(resolveAgentModelPatterns({ agentModel: "@smol", settings })).toEqual(["anthropic/claude-sonnet-4-5"]); }); test("prefers configured designer role override over priority defaults", () => { @@ -809,7 +810,7 @@ describe("resolveAgentModelPatterns", () => { }); const result = resolveAgentModelPatterns({ - agentModel: "pi/designer", + agentModel: "@designer", settings, }); @@ -818,7 +819,7 @@ describe("resolveAgentModelPatterns", () => { test("slow priority falls forward to Opus 4.8 before older Opus aliases", () => { const settings = Settings.isolated(); - const patterns = resolveAgentModelPatterns({ agentModel: "pi/slow", settings }); + const patterns = resolveAgentModelPatterns({ agentModel: "@slow", settings }); const dottedRegistry = { getAvailable: () => [ @@ -898,6 +899,70 @@ describe("resolveCliModel", () => { expect(result.model?.id).toBe("gpt-4o"); }); + test("resolves configured custom, legacy, and default role aliases from --model", () => { + const registry = { + getAll: () => allModels, + }; + const settings = Settings.isolated({ + modelRoles: { + default: "openai/gpt-4o", + fable: "anthropic/claude-sonnet-4-5:high", + }, + }); + + const canonical = resolveCliModel({ + cliModel: "@fable", + modelRegistry: registry, + settings, + }); + const legacy = resolveCliModel({ + cliModel: `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}fable`, + modelRegistry: registry, + settings, + }); + const defaultRole = resolveCliModel({ + cliModel: DEFAULT_MODEL_ROLE_ALIAS, + modelRegistry: registry, + settings, + }); + + expect(canonical.error).toBeUndefined(); + expect(canonical.model?.provider).toBe("anthropic"); + expect(canonical.model?.id).toBe("claude-sonnet-4-5"); + expect(canonical.thinkingLevel).toBe(Effort.High); + expect(legacy).toEqual(canonical); + expect(defaultRole.model?.provider).toBe("openai"); + expect(defaultRole.model?.id).toBe("gpt-4o"); + }); + + test("splits thinking suffixes and abbreviations off the * default alias", () => { + const registry = { + getAll: () => allModels, + }; + const settings = Settings.isolated({ + modelRoles: { default: "anthropic/claude-sonnet-4-5" }, + }); + + const explicit = resolveCliModel({ + cliModel: `${DEFAULT_MODEL_ROLE_ALIAS}:high`, + modelRegistry: registry, + settings, + }); + const abbreviated = resolveCliModel({ + cliModel: `${DEFAULT_MODEL_ROLE_ALIAS}:xhi`, + modelRegistry: registry, + settings, + }); + + expect(explicit.error).toBeUndefined(); + expect(explicit.model?.id).toBe("claude-sonnet-4-5"); + expect(explicit.thinkingLevel).toBe(Effort.High); + // `xhi` → xhigh via unique-prefix parsing, then clamped to the model ladder. + expect(abbreviated.error).toBeUndefined(); + expect(abbreviated.model?.id).toBe("claude-sonnet-4-5"); + expect(abbreviated.thinkingLevel).toBe(Effort.High); + }); + test("resolves fuzzy patterns within an explicit provider", () => { const registry = { getAll: () => allModels, @@ -1108,6 +1173,25 @@ describe("resolveModelScope", () => { expect(scoped[0].model.id).toBe("gpt-5.5"); }); + test("resolves role aliases in --models scope to the role's model with its thinking level", async () => { + const settings = Settings.isolated({ + modelRoles: { fable: "anthropic/claude-sonnet-4-5:high" }, + }); + + const scoped = await resolveModelScope( + ["@fable", "openai/gpt-4o"], + { getAvailable: () => allModels }, + undefined, + settings, + ); + + expect(scoped).toHaveLength(2); + expect(scoped[0].model.id).toBe("claude-sonnet-4-5"); + expect(scoped[0].thinkingLevel).toBe(Effort.High); + expect(scoped[0].explicitThinkingLevel).toBe(true); + expect(scoped[1].model.id).toBe("gpt-4o"); + }); + test("applies max thinking selectors to glob scopes when no literal max ids match", async () => { const registry = { getAvailable: () => mockCodexOverlapModels, @@ -1272,18 +1356,18 @@ describe("resolveModelFromString", () => { }); describe("expandRoleAlias", () => { - test("expands pi/vision to configured vision role", () => { + test("expands @vision to configured vision role", () => { const settings = Settings.isolated(); settings.setModelRole("vision", "openai/gpt-4o"); - expect(expandRoleAlias("pi/vision", settings)).toBe("openai/gpt-4o"); + expect(expandRoleAlias("@vision", settings)).toBe("openai/gpt-4o"); }); - test("keeps pi/vision alias when vision role is unset", () => { + test("keeps @vision alias when vision role is unset", () => { const settings = Settings.isolated(); settings.setModelRole("default", "anthropic/claude-sonnet-4-5"); - expect(expandRoleAlias("pi/vision", settings)).toBe("pi/vision"); + expect(expandRoleAlias("@vision", settings)).toBe("@vision"); }); }); @@ -1305,7 +1389,7 @@ describe("extractExplicitThinkingSelector", () => { test("treats max on pi role aliases as an explicit selector before expansion", () => { const settings = Settings.isolated(); settings.setModelRole("smol", "nanogpt/coding-router:max"); - const result = extractExplicitThinkingSelector("pi/smol:max", settings, { + const result = extractExplicitThinkingSelector("@smol:max", settings, { isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); expect(result).toBe(Effort.Max); @@ -1442,6 +1526,15 @@ describe("filterAvailableModelsByEnabledPatterns", () => { expect(filterAvailableModelsByEnabledPatterns(models, [])).toEqual(models); }); + test("resolves role aliases to the role's model when settings are provided", () => { + const settings = Settings.isolated({ + modelRoles: { fable: "anthropic/claude-sonnet-4-5:high" }, + }); + const result = filterAvailableModelsByEnabledPatterns(models, ["@fable"], settings); + expect(result).toHaveLength(1); + expect(result[0].id).toBe("claude-sonnet-4-5"); + }); + test("filters by exact provider/modelId", () => { const result = filterAvailableModelsByEnabledPatterns(models, ["anthropic/claude-sonnet-4-5"]); expect(result).toHaveLength(1); diff --git a/packages/coding-agent/test/role-thinking-helper-propagation.test.ts b/packages/coding-agent/test/role-thinking-helper-propagation.test.ts index 882d20644..9310e2833 100644 --- a/packages/coding-agent/test/role-thinking-helper-propagation.test.ts +++ b/packages/coding-agent/test/role-thinking-helper-propagation.test.ts @@ -39,7 +39,7 @@ describe("role thinking helper propagation", () => { const model = getModelOrThrow("claude-sonnet-4-5"); const settings = createSettings({ default: `${model.provider}/${model.id}:high`, - smol: "pi/default:minimal", + smol: "@default:minimal", }); const registry = { getAvailable: () => [model], @@ -85,7 +85,7 @@ describe("role thinking helper propagation", () => { const model = getModelOrThrow("claude-sonnet-4-5"); const settings = createSettings({ default: `${model.provider}/${model.id}:high`, - smol: "pi/default:low", + smol: "@default:low", }); const registry = { getAvailable: () => [model], diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index e2706d95e..278ef70b0 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -213,7 +213,7 @@ describe("createAgentSession deferred model pattern resolution", () => { settings.setModelRole("smol", "runtime-provider/runtime-model"); const { session, modelFallbackMessage } = await createAgentSession({ - ...(await buildSessionOptions("pi/smol")), + ...(await buildSessionOptions("@smol")), settings, }); @@ -265,7 +265,7 @@ describe("createAgentSession deferred model pattern resolution", () => { test("does not apply default role thinking override when modelPattern is explicit", async () => { const settings = Settings.isolated({ defaultThinkingLevel: "off" }); settings.setModelRole("smol", "runtime-provider/runtime-reasoning-model"); - settings.setModelRole("default", "pi/smol:high"); + settings.setModelRole("default", "@smol:high"); const { session } = await createAgentSession({ ...(await buildSessionOptions("runtime-provider/runtime-reasoning-model")), diff --git a/packages/coding-agent/test/task/executor-pass-through.test.ts b/packages/coding-agent/test/task/executor-pass-through.test.ts index 642d799ca..7a79547a1 100644 --- a/packages/coding-agent/test/task/executor-pass-through.test.ts +++ b/packages/coding-agent/test/task/executor-pass-through.test.ts @@ -175,7 +175,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const result = await runSubprocess({ ...baseOptions, - agent: { ...baseAgent, model: ["pi/task"] }, + agent: { ...baseAgent, model: ["@task"] }, id: "subagent-thinking-precedence", settings, modelRegistry: createModelRegistry(model), @@ -199,7 +199,7 @@ describe("runSubprocess parent-discovery pass-through (issue #2190)", () => { const result = await runSubprocess({ ...baseOptions, - agent: { ...baseAgent, model: ["pi/task"] }, + agent: { ...baseAgent, model: ["@task"] }, id: "subagent-thinking-default", settings, modelRegistry: createModelRegistry(model), diff --git a/packages/coding-agent/test/tools/inspect-image.test.ts b/packages/coding-agent/test/tools/inspect-image.test.ts index 91dfcc661..bfdced519 100644 --- a/packages/coding-agent/test/tools/inspect-image.test.ts +++ b/packages/coding-agent/test/tools/inspect-image.test.ts @@ -352,7 +352,7 @@ describe("InspectImageTool", () => { expect(stub.calls).toHaveLength(0); }); - it("falls back to pi/default when vision role is unset", async () => { + it("falls back to @default when vision role is unset", async () => { const imagePath = path.join(testDir, "screen.png"); fs.writeFileSync(imagePath, Buffer.from(TINY_PNG_BASE64, "base64")); From da24614d5afdb4b328a92da1eeb816f5a451fb5d Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 14 Jul 2026 00:39:34 +0200 Subject: [PATCH 23/28] feat: added scrollback rebuild controls and prewalk status-line visibility - Added `tui.scrollbackRebuild` configuration with interactive startup/controller wiring to apply `setScrollbackRebuild`. - Exposed prewalk session state in `SegmentContext` and rendered a dedicated prewalk segment/icon in the status line. - Added divergence-aware TUI full-paint logic that enables scrollback erase-and-replay rebuilds for non-multiplexer divergence cases. - Updated rendering and streaming tests to verify rebuild behavior (`3J`) and eliminate stale marker expectations under drift scenarios. --- packages/agent/CHANGELOG.md | 8 +- packages/ai/CHANGELOG.md | 15 +- packages/coding-agent/CHANGELOG.md | 67 ++--- .../src/config/settings-schema.ts | 11 + .../components/status-line/component.test.ts | 43 ++- .../modes/components/status-line/component.ts | 4 + .../modes/components/status-line/segments.ts | 6 + .../src/modes/components/status-line/types.ts | 3 + .../modes/controllers/selector-controller.ts | 4 + .../src/modes/interactive-mode.ts | 1 + .../coding-agent/src/modes/theme/theme.ts | 5 + .../coding-agent/src/session/agent-session.ts | 5 + .../test/status-line-model.test.ts | 1 + .../test/status-line-overflow.test.ts | 1 + .../test/status-line-path.test.ts | 1 + .../test/status-line-time-spent.test.ts | 1 + packages/hashline/CHANGELOG.md | 6 +- packages/snapcompact/CHANGELOG.md | 2 +- packages/tui/CHANGELOG.md | 6 +- packages/tui/src/tui.ts | 88 ++++-- packages/tui/test/render-regressions.test.ts | 264 +++++++----------- .../test/streaming-scrollback-defer.test.ts | 186 ++++++++---- 22 files changed, 418 insertions(+), 310 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index ba7b65aec..89ca1813b 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -4,12 +4,8 @@ ### Added -- Added automated image-dropping rescue tier to compaction dead-end recovery -- Added visual warnings to the session timeline when compaction fails to free sufficient space - -### Changed - -- Improved compaction dead-end notifications with specific recovery instructions +- Added an automated image-dropping rescue tier to compaction dead-end recovery. +- Added visual warnings and detailed recovery instructions to the session timeline when compaction fails to free sufficient space. ## [16.4.5] - 2026-07-11 diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 712cd2e4f..5199da7c9 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,20 +4,19 @@ ### Added -- Added diagnostic response headers to auth-gateway inference endpoints: `x-request-id`/`request-id` (correlates with gateway logs; surfaced by OpenAI/Anthropic SDKs) and LiteLLM-style `x-litellm-model-id`/`x-litellm-model-api-base` on every response, plus `x-litellm-response-cost`, `x-litellm-response-duration-ms`, and `openai-processing-ms` on non-streaming responses +- Added diagnostic response headers to auth-gateway inference endpoints, including request IDs (x-request-id/request-id), LiteLLM model metadata (x-litellm-model-id/x-litellm-model-api-base), and performance/cost metrics (x-litellm-response-cost, x-litellm-response-duration-ms, openai-processing-ms) on non-streaming responses. ### Changed -- Switched Google and Google Vertex providers to always use `streamGenerateContent` requests - -### Removed - -- Removed automatic `/interactions` chaining for follow-up turns in Google provider calls -- Removed `useInteractionsApi`, `storeInteraction`, and `previousInteractionId` from stream options +- Updated Google and Google Vertex providers to always use streamGenerateContent requests. ### Fixed -- Fixed empty provider responses (e.g. "Cloud Code Assist API returned an empty response") being classified as non-retryable: `ProviderResponseError` with kind `empty-body` now carries the transient flag, so session retry and configured model-fallback chains engage instead of hard-failing the turn +- Fixed empty provider responses (such as from Cloud Code Assist API) being classified as non-retryable, allowing session retries and model-fallback chains to engage instead of failing the turn. + +### Removed + +- Removed automatic /interactions chaining for follow-up turns in Google provider calls, along with the useInteractionsApi, storeInteraction, and previousInteractionId stream options. ## [16.4.6] - 2026-07-12 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8175722cf..1e3b511fa 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,54 +4,47 @@ ### Breaking Changes -- Replaced the `--reasoning-slide-*` flag family (`--reasoning-slide-model`, `--reasoning-slide-turns`, `--reasoning-slide-on-action`, `--reasoning-slide-plan`, `--reasoning-slide-plan-at`, `--reasoning-slide-checklist`) with a single prewalk mechanism: `--prewalk` switches from the starting model to a fast/cheap target at the first completed turn that starts execution — the todo-list init the plan nudge asks for, or any edit/write tool — always with the hidden plan nudge before the switch and the verify-before-finishing checklist after it (the configuration that won benchmark testing). `--prewalk-into ` overrides the default "smol"-role target and implies `--prewalk`; `--no-prewalk` force-disables. The fixed-turn trigger and per-piece plan/checklist toggles are gone. +- Replaced the `--reasoning-slide-*` flag family with a unified `--prewalk` mechanism (`--prewalk`, `--prewalk-into `, and `--no-prewalk`) to manage model handoffs during execution. ### Added -- Added `--prewalk` / `--prewalk-into ` / `--no-prewalk`: start on a strong model, then hand off to a fast/cheap one (default the `smol` role) at the first edit/write tool call *after* the todo list has been initialized. The starting model handles all planning and todo initialization, and begins implementation, before handing off; the fast model includes a verify checklist before finishing. Enable per-user with the `prewalk.enabled` setting; force it mid-session with the new `/prewalk` slash command, which arms the switch. -- Added display setting to toggle between collapsing or keeping compacted history inline, now applied to live session displays -- Added a compact session-only model picker (Alt+P) for quick model switching without changing roles -- Added `@` search to the Alt+P / `/switch` picker: it lists configured Ctrl+P quick roles in matching segment colors and applies the selected role's model and thinking for the current session. -- Redesigned Agent Hub entries as two-line cards: identity (status glyph, name, agent type, parent when nested) on the left, active model + reasoning level and age right-aligned, with the task description on its own line; dropped the redundant `sub · of Main` noise -- Added a project-scoped `launch` tool for shared long-running services and debuggers, with readiness probes, bounded logs, PTY input, restart policies, and automatic teardown after the last omp instance exits. Gated behind the `launch.enabled` setting (default on); when disabled the tool is withdrawn and the bash prompt drops its "use launch" guidance. -- Added `detached` `launch` starts for standalone services that survive every omp instance and broker shutdown, then reconnect to the next broker for logs and explicit stop. +- Added a new `--prewalk` execution flow (with `--prewalk-into ` and `--no-prewalk` overrides) that starts tasks on a strong model for planning and todo initialization before handing off to a faster, cheaper model for implementation. + - Added a status line annotation for the active prewalk phase (armed or active). + - Added the `tui.scrollbackRebuild` setting to gate the erase-and-replay native scrollback rebuild mechanism (defaults to off). +- Added a display setting to toggle between collapsing or keeping compacted history inline in live session displays. +- Added a compact session-only model picker (Alt+P) for quick model switching, featuring `@` search to quickly list and apply configured quick roles. +- Redesigned Agent Hub entries into a cleaner two-line card layout showing identity, active model, reasoning level, age, and task description. +- Added a project-scoped `launch` tool (gated by `launch.enabled`) for managing shared long-running services and debuggers, featuring readiness probes, bounded logs, PTY input, restart policies, and automatic teardown. +- Added support for `detached` launches, allowing standalone services to survive broker shutdowns and reconnect to subsequent sessions. ### Changed -- Updated `--mode json` logs to include provider payloads in auto-compaction events -- Refined prewalk planning to require 5-9 meaningful todo items with concrete targets and checks -- Updated tangential agent forks to ignore parent session history and focus exclusively on the new request -- Hardened `/tan` fork isolation: the clone's inherited todo list is cleared at fork (parent todo reminders no longer drag the tan back onto the parent's task), the fork notice warns that the parent is concurrently editing the same working directory, and the notice is re-injected after each compaction so the fork boundary survives summarization -- Added visual markers in the transcript for elided tool calls that have no corresponding result -- Updated status event log to prioritize the most recent entries in the display window -- Updated the snapcompact shape preview transcript to use the compact scope format shown to models during compaction. -- Bumped `@agentclientprotocol/sdk` 0.25.0 → 1.2.1 (major); patched the package's `exports` map to restore the `dist/schema/zod.gen.js` subpath the SDK no longer publishes, which our tests import for response-shape validation. - -### Removed - -- Removed the `--prewalk-boomerang` feature and its associated configuration setting -- Removed the unreliable Bing and Yahoo HTML-scraping web search providers +- Updated JSON logs (`--mode json`) to include provider payloads in auto-compaction events. +- Updated tangential agent forks (`/tan`) to ignore parent session history and focus exclusively on the new request, hardening isolation with cleared todo lists and concurrent editing warnings. +- Added visual markers in the transcript for elided tool calls that have no corresponding result. +- Updated the status event log to prioritize the most recent entries in the display window. +- Upgraded `@agentclientprotocol/sdk` to version 1.2.1. ### Fixed -- Fixed expanded (ctrl+O) streaming edit previews duplicating the tool box in terminal scrollback: an unbounded live diff scrolled above the native-scrollback commit boundary mid-stream, freezing a stale preview snapshot that the finalized render then recommitted below. Expanded previews now use a viewport-sized tail window; the full diff still renders once the result finalizes. -- Fixed configured custom model roles resolving from `--model`; canonical role selectors now use `@role`, legacy selectors remain supported, and `*` selects the default role. Thinking suffixes split correctly off the bare `*` alias (`*:xhigh`), thinking selectors accept unambiguous abbreviations (`:xhi`, `:med`) on every surface, and role aliases now also work in `--models` and `enabledModels` scopes (each role contributes its resolved model). -- Fixed quadratic growth of `--mode json` logs by eliding redundant message snapshots and payloads -- Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache -- Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan -- Fixed inconsistent history rendering when toggling the display setting for compacted items -- Fixed configured `retry.fallbackChains` never engaging on non-retryable provider errors (e.g. "Cloud Code Assist API returned an empty response"): a hard error on a model covered by a fallback chain now switches to the next candidate instead of failing the turn, while still never backoff-retrying the failing model itself -- Fixed transcript rebuilds (compaction, `/compact`, and toggling history display) repainting content below stale scrollback when collapsing history; rebuilds now correctly clear the scrollback buffer when history is collapsed -- Improved auto-compaction to automatically drop images and elide content when context is tight, and added persistent warning badges to the compaction divider when manual intervention is required -- Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately -- Fixed the prewalk plan nudge silently ending the run with no code written when the model answered with a text-only reply (no tool call): the agent loop treats a tool-call-free turn as a natural stop and never prompts again, which the nudge's own "write the plan in your next reply" instruction makes common. The nudge now explicitly tells the model this is a checkpoint, not a final answer, and the session forces one more turn whenever a post-nudge reply lands with zero tool calls -- Fixed launch tool rendering stacking a stale pending header over a bare `✓ Launch` line and raw text: the tool now uses a merged registry renderer with one per-op status header (op, target, `state · pid · uptime` meta), stripped log cursor suffixes, capped collapsed log/list previews, and a launch tool glyph -- Fixed `launch logs` flattening PTY control sequences into repeated or diagonally wrapped debugger fragments: the bounded raw PTY stream is now replayed before row selection through the shared xterm screen renderer used by Bash PTY mode, blank terminal cells retain their columns, and the final colored/styled viewport renders in the same bordered output block as Bash while model-facing text remains sanitized -- Fixed confusing launch start/wait results when readiness timed out with the log pattern already matched (readiness needs log AND port): the result printed a contradictory `Ready: ` next to `Readiness timed out` without naming the failing condition. Daemon snapshots now carry the unmet conditions (`readyPending`), and start/wait results state exactly what never happened (e.g. `port 3100 on 127.0.0.1 never accepted connections`); the TUI shows a `waiting on port` badge on starting daemons -- Fixed the in-process `stat` builtin mangling BSD-style invocations like `stat -f "%Sm %N" file` (macOS muscle memory): GNU `-f` means `--file-system`, so the format string was treated as a file operand — printing filesystem info for the real operands and erroring with `cannot read file system information for '%Sm %N'`. A `-f` whose format value contains `%` is now detected as BSD syntax and translated to the GNU equivalent (`%Sm`→`%y`, `%N`→`%n`, `%z`→`%s`, epoch/`S`-form times, owner/group/permission and `H`/`L` sub-field directives, `-L`/`-n`/`-q`/`-F` flag clusters, with `%n`/`%t` as literal newline/tab); directives with no GNU counterpart fail with a clear `unsupported BSD format directive` error -- Fixed the remaining GNU-flavored shell builtins that broke under macOS/BSD muscle memory, using the same unambiguous-detection approach as the `stat` fix (only invocations that are invalid or nonsensical under GNU semantics are reinterpreted; unsupported BSD forms fail loudly instead of producing wrong output): `date -r ` formats the epoch when no such file exists (GNU `-r FILE` mtime preserved), signed `date -v±N` adjustments translate to `-d` relative dates and `-j` is accepted (`-j -f` strptime parse mode and field-set `-v` error clearly); `sed -i '' 's/…/…/' file` drops the BSD empty backup-suffix token instead of treating it as the script; `mktemp -t prefix` without X's creates `$TMPDIR/prefix.XXXXXXXXXX` (the GNU `too few X's` error path); `tail -r` reverses input by delegating to `tac` (with `-n`/`-c`/`-f` combinations erroring clearly); `find -E` maps to `-regextype posix-extended` ahead of the expression; `base64 -D` decodes as an alias of `-d`; and `ln -sfh` works via a `-h` alias of `--no-dereference` (clap's `-h` help short is dropped to match real GNU/BSD ln; `--help` unchanged) +- Fixed terminal scrollback duplication issues with expanded streaming edit previews (Ctrl+O) by using a viewport-sized tail window. +- Fixed custom model role resolution and alias parsing, ensuring canonical role selectors (`@role`) and thinking suffixes resolve correctly across all configuration surfaces. +- Fixed quadratic growth in JSON logs by eliding redundant message snapshots and payloads. +- Fixed prompt cache misses and incorrect cache key pinning for `/tan` and `/fork` clones. +- Fixed inconsistent history rendering and scrollback repainting when toggling the display setting for compacted items. +- Fixed `retry.fallbackChains` failing to engage on non-retryable provider errors, ensuring the agent correctly falls back to the next candidate model. +- Improved auto-compaction to automatically drop images and elide content when context is tight, and added persistent warning badges when manual intervention is required. +- Fixed backgrounded Bash blocks continuing to repaint with live output; they now freeze with a compact job notice while completion is delivered separately. +- Fixed rendering, status display, and PTY control sequence formatting issues in the `launch` tool. +- Fixed in-process shell builtins (including `stat`, `date`, `sed`, `mktemp`, `tail`, `find`, `base64`, and `ln`) to correctly detect and translate macOS/BSD-style arguments and flags, preventing failures caused by GNU-only assumptions. + +### Removed + +- Removed the `--prewalk-boomerang` feature and its associated configuration setting. +- Removed the unreliable Bing and Yahoo HTML-scraping web search providers. ## [16.4.8] - 2026-07-12 + ### Added - Added a predicate form to the browser run's `wait()` helper: `wait(fn, { timeout?, interval? })` polls the function (sync or async) until truthy and resolves with that value, failing with a named timeout error (deadline clamped under the cell budget so it always beats the opaque whole-cell timeout) instead of Bun's `sleep expects a number` or a whole-cell stall from in-page polling Promises; both `wait` forms now register in the stall diagnosis of cell timeouts diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index aa08b363d..b5be79ab8 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -897,6 +897,17 @@ export const SETTINGS_SCHEMA = { description: "Remove the 1-character horizontal padding from the left and right of the terminal output", }, }, + "tui.scrollbackRebuild": { + type: "boolean", + default: false, + ui: { + tab: "appearance", + group: "Display", + label: "Rewrite Scrollback", + description: + "Erase and replay terminal scrollback when a block's final form replaces its live preview. When off (default), stale preview copies remain in history and the final content is appended below.", + }, + }, "display.shimmer": { type: "enum", diff --git a/packages/coding-agent/src/modes/components/status-line/component.test.ts b/packages/coding-agent/src/modes/components/status-line/component.test.ts index 559a11870..0f5efdeb9 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.test.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.test.ts @@ -4,15 +4,43 @@ import type { AgentSession } from "../../../session/agent-session"; import { getThemeByName, setThemeInstance } from "../../theme/theme"; import { StatusLineComponent } from "./component"; -function makeSessionWithLastMessage(lastMessage: unknown) { +function makeSessionWithLastMessage(lastMessage: unknown, prewalkArmed: boolean = false) { return { - messages: [lastMessage], + messages: lastMessage ? [lastMessage] : [], model: { contextWindow: 128000 }, contextUsageRevision: 0, systemPrompt: [], agent: { state: { tools: [] } }, skills: [], getContextUsage: () => ({ tokens: 42, contextWindow: 128000 }), + state: { + messages: lastMessage ? [lastMessage] : [], + model: { contextWindow: 128000 }, + }, + sessionManager: { + getUsageStatistics: () => ({ + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + orchestrationInput: 0, + orchestrationOutput: 0, + orchestrationCacheRead: 0, + premiumRequests: 0, + cost: 0, + tokensPerSecond: null, + }), + getSessionName: () => "test-session", + }, + getPrewalkState: () => (prewalkArmed ? { target: { id: "cheap-model", provider: "openai" } } : undefined), + getAsyncJobSnapshot: () => undefined, + isAdvisorActive: () => false, + isFastModeActive: () => false, + configuredThinkingLevel: () => undefined, + modelRegistry: { + isUsingOAuth: () => false, + }, }; } @@ -41,4 +69,15 @@ describe("StatusLineComponent", () => { expect(statusLine.getCachedContextBreakdown()).toEqual({ usedTokens: 42, contextWindow: 128000 }); }); + + it("renders Prewalk annotation when prewalk is armed", () => { + const statusLine = new StatusLineComponent(makeSessionWithLastMessage(null, true) as unknown as AgentSession); + + // By default preset, 'mode' segment is included in left/right segments. + // Let's get the border and see if Prewalk is rendered. + const border = statusLine.getTopBorder(100); + // SGR codes might be included, so we check if the stripped content contains "Prewalk" + const stripped = border.content.replace(/\x1b\[[0-9;]*m/g, ""); + expect(stripped).toContain("Prewalk"); + }); }); diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index e709f9d78..f07f30e86 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -1051,6 +1051,10 @@ export class StatusLineComponent implements Component { compactThinkingLevel: this.#resolveSettings().compactThinkingLevel ?? false, planMode: this.#planModeStatus, loopMode: this.#loopModeStatus, + prewalk: + typeof this.session.getPrewalkState === "function" && this.session.getPrewalkState() + ? { enabled: true } + : null, goalMode: this.#goalModeStatus, vibeMode: this.#vibeModeStatus, collab: this.#collabStatus, diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts index e1ef2b87e..079103426 100644 --- a/packages/coding-agent/src/modes/components/status-line/segments.ts +++ b/packages/coding-agent/src/modes/components/status-line/segments.ts @@ -208,6 +208,12 @@ const modeSegment: StatusLineSegment = { return { content: theme.fg(color, content), visible: true }; } + const prewalk = ctx.prewalk; + if (prewalk?.enabled) { + const content = withIcon(theme.icon.prewalk, "Prewalk"); + return { content: theme.fg("accent", content), visible: true }; + } + const goal = ctx.goalMode; if (goal && (goal.enabled || goal.paused)) { return renderGoalMode(ctx, goal); diff --git a/packages/coding-agent/src/modes/components/status-line/types.ts b/packages/coding-agent/src/modes/components/status-line/types.ts index 8fe16380d..06719799b 100644 --- a/packages/coding-agent/src/modes/components/status-line/types.ts +++ b/packages/coding-agent/src/modes/components/status-line/types.ts @@ -60,6 +60,9 @@ export interface SegmentContext { enabled: boolean; paused: boolean; } | null; + prewalk: { + enabled: boolean; + } | null; loopMode: { enabled: boolean; } | null; diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index adbd74d38..d4fe93c91 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -466,6 +466,10 @@ export class SelectorController { this.ctx.ui.requestRender(); break; + case "tui.scrollbackRebuild": + this.ctx.ui.setScrollbackRebuild(value as boolean); + break; + case "tui.renderMermaid": setMarkdownMermaidRendering(value as boolean); this.ctx.session.refreshBaseSystemPrompt().catch(err => { diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8311acbc2..50cba2279 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -665,6 +665,7 @@ export class InteractiveMode implements InteractiveModeContext { setMarkdownMermaidRendering(settings.get("tui.renderMermaid")); this.ui = new TUI(new ProcessTerminal(), settings.get("showHardwareCursor")); this.ui.setMaxInlineImages(settings.get("tui.maxInlineImages")); + this.ui.setScrollbackRebuild(settings.get("tui.scrollbackRebuild")); // OSC 66 text-sizing is Kitty-only; resolve the setting against the terminal's // capability (`TERMINAL.textSizing` defaults on for Kitty) so it stays off // unless the user opts in, and never emits raw escapes on other terminals. diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index e6ae57aa4..3c64156fc 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -92,6 +92,7 @@ export type SymbolKey = // Icons | "icon.model" | "icon.plan" + | "icon.prewalk" | "icon.goal" | "icon.pause" | "icon.loop" @@ -301,6 +302,7 @@ const UNICODE_SYMBOLS: SymbolMap = { // Icons "icon.model": "⬢", "icon.plan": "🗺", + "icon.prewalk": "🏃", "icon.goal": "🎯", "icon.pause": "⏸", "icon.loop": "↻", @@ -562,6 +564,7 @@ const NERD_SYMBOLS: SymbolMap = { "icon.model": "\uec19", // pick:  | alt:   "icon.plan": "\uf2d2", + "icon.prewalk": "\uf29d", // pick: (nf-fa-bullseye) | alt: (nf-md-target) ◎ ⌖ "icon.goal": "\uf140", // pick: (nf-fa-pause) | alt: ⏸ || @@ -819,6 +822,7 @@ const ASCII_SYMBOLS: SymbolMap = { // Icons "icon.model": "[M]", "icon.plan": "plan", + "icon.prewalk": "prewalk", "icon.goal": "goal", "icon.pause": "||", "icon.loop": "loop", @@ -1819,6 +1823,7 @@ export class Theme { return { model: this.#symbols["icon.model"], plan: this.#symbols["icon.plan"], + prewalk: this.#symbols["icon.prewalk"], goal: this.#symbols["icon.goal"], pause: this.#symbols["icon.pause"], loop: this.#symbols["icon.loop"], diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index c6bc8b20b..dcc716744 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -7473,6 +7473,11 @@ export class AgentSession { return this.#planModeState; } + /** Prewalk state, if armed and active */ + getPrewalkState(): Prewalk | undefined { + return this.#prewalk; + } + setPlanModeState(state: PlanModeState | undefined): void { this.#planModeState = state; if (state?.enabled) { diff --git a/packages/coding-agent/test/status-line-model.test.ts b/packages/coding-agent/test/status-line-model.test.ts index 6a2e5ebd0..43af37276 100644 --- a/packages/coding-agent/test/status-line-model.test.ts +++ b/packages/coding-agent/test/status-line-model.test.ts @@ -22,6 +22,7 @@ function createModelContext(advisorActive: boolean): SegmentContext { options: {}, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-overflow.test.ts b/packages/coding-agent/test/status-line-overflow.test.ts index 996389989..a4a860e00 100644 --- a/packages/coding-agent/test/status-line-overflow.test.ts +++ b/packages/coding-agent/test/status-line-overflow.test.ts @@ -45,6 +45,7 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null }, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-path.test.ts b/packages/coding-agent/test/status-line-path.test.ts index c32526a59..a238b606e 100644 --- a/packages/coding-agent/test/status-line-path.test.ts +++ b/packages/coding-agent/test/status-line-path.test.ts @@ -31,6 +31,7 @@ function createPathContext(): SegmentContext { }, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/coding-agent/test/status-line-time-spent.test.ts b/packages/coding-agent/test/status-line-time-spent.test.ts index 35620a132..e5dd3df92 100644 --- a/packages/coding-agent/test/status-line-time-spent.test.ts +++ b/packages/coding-agent/test/status-line-time-spent.test.ts @@ -41,6 +41,7 @@ function createCtx(activeMs: number): SegmentContext { options: {}, planMode: null, loopMode: null, + prewalk: null, goalMode: null, vibeMode: null, collab: null, diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index 897f05038..e137e771c 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -4,9 +4,9 @@ ### Fixed -- Rejected ambiguous swaps that risk silent deletion of range boundaries -- Prevented ambiguous auto-repairing of structural closing lines when payload placement is unclear -- Prevented stale-hash recovery from relocating edits onto duplicated context after the original target changed +- Fixed a critical issue where ambiguous swaps could silently delete range boundaries. +- Prevented incorrect auto-repairing of structural closing lines when payload placement is ambiguous. +- Fixed a bug in stale-hash recovery that could incorrectly relocate edits onto duplicated context after the original target changed. ## [16.3.3] - 2026-07-02 diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index d020535e9..5e2295ee4 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- Changed archived transcript rendering to compact `¶user:`, `¶think:`, `¶ai:`, and `¶call:` scopes; repeated adjacent scopes now continue as plain lines, tool-call intents trail calls as `//` comments, and the compaction prompt documents the format. +- Updated archived transcript rendering to use a more compact format with `¶user:`, `¶think:`, `¶ai:`, and `¶call:` scopes, omitting repeated adjacent scope headers and appending tool-call intents as comments. ## [16.3.7] - 2026-07-05 diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 0007839a3..3c61b032d 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,9 +2,13 @@ ## [Unreleased] +### Changed + + - Improved native scrollback history management by introducing an optional erase-and-replay mechanism to rebuild scrollback when mutated rows (such as finalized tool blocks or collapsed transcripts) diverge. This is now gated behind the `tui.scrollbackRebuild` setting and defaults to off. + ### Fixed -- Fixed forced renders (tool finalization, `resetDisplay`, image reconciliation) landing during a resize drag preempting the alternate-screen viewport fast path: each one left the borrowed alt screen, erased native scrollback (ED3), and visibly replayed the whole transcript on the normal screen mid-drag — then the settle replayed it again. Forced intent now folds into the single authoritative settle paint. +- Fixed a rendering issue where resizing the terminal during forced renders (such as tool finalization or image reconciliation) caused the entire transcript to visibly replay and flicker. Forced renders are now consolidated into a single paint once the resize settles. ## [16.4.7] - 2026-07-12 diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 56206cc61..47668a632 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -5,11 +5,15 @@ * immutable — the tape is the terminal's visual record. Whatever scrolls * above the window enters history exactly once, in order: as exact-final * bytes when the component seam (`NativeScrollbackLiveRegion`) declared them - * final, else as a frozen snapshot of what was on screen. ED3 (`CSI 3 J`) is - * emitted only for gesture-driven replays (session replace, resize, - * resetDisplay) where snapping the viewport is acceptable. The engine never - * probes or guesses the terminal's scroll position, and the hot path clamps - * over-wide lines instead of throwing. See `docs/tui-core-renderer.md`. + * final, else as a frozen snapshot of what was on screen. When recorded + * history diverges from the frame (a finalized block replacing its + * scrolled-off live render), the engine erases and replays (ED3, `CSI 3 J`) + * so history holds the content exactly once — the same replay used for + * gestures (session replace, resize, resetDisplay). Multiplexer panes, where + * ED3 is unsafe, instead re-anchor and recommit below the stale fragment — + * duplication, never loss. The engine never probes or guesses the terminal's + * scroll position, and the hot path clamps over-wide lines instead of + * throwing. See `docs/tui-core-renderer.md`. */ import * as fs from "node:fs"; import { performance } from "node:perf_hooks"; @@ -783,9 +787,11 @@ const RESYNC_TAIL_SAMPLES = 8; * source just became declared-final (the block finalized / a barrier * cleared). Hard-scanned in FULL with no tolerance: any content change * (a pending header settling, a preview replaced by its result, a tail - * shifting up after a barrier removal) re-anchors so the final content - * recommits below the frozen snapshot — duplication, never loss — - * instead of being committed nowhere and painted nowhere. + * shifting up after a barrier removal) re-anchors so the engine can + * erase-and-replay history with the final content exactly once (or, on + * ED3-unsafe multiplexers, recommit it below the frozen snapshot — + * duplication, never loss) instead of committing it nowhere and + * painting it nowhere. * [finalTo, prefix.length) FROZEN visual snapshots of still-live rows — * exempt: their drift is expected (a collapsing preview, a ticking * progress tree) and must never spray re-anchors mid-run. @@ -1001,6 +1007,8 @@ export class TUI extends Container { #clearScrollbackOnNextRender = false; #forceViewportRepaintOnNextRender = false; #hasEverRendered = false; + #scrollbackRebuildEnabled = + Bun.env.PI_TUI_SCROLLBACK_REBUILD === "1" || Bun.env.PI_TUI_SCROLLBACK_REBUILD === "true"; // Set by the terminal resize callback; consumed by the next render. A resize // event invalidates the committed screen even when the dimensions net out // unchanged by render time (e.g. a 6→4→6 round trip coalesced into one frame @@ -1290,6 +1298,23 @@ export class TUI extends Container { this.#imageBudget.setCap(cap); } + /** + * Get whether scrollback divergence rebuild is enabled. + */ + getScrollbackRebuild(): boolean { + return this.#scrollbackRebuildEnabled; + } + + /** + * Enable or disable scrollback divergence rebuild (default off). + * When enabled, the engine will erase and replay the terminal's + * scrollback (using ED3 / alt buffer / scrollback replay) to avoid + * duplicate blocks when a block's final form replaces its live preview. + */ + setScrollbackRebuild(enabled: boolean): void { + this.#scrollbackRebuildEnabled = enabled; + } + getShowHardwareCursor(): boolean { return this.#showHardwareCursor; } @@ -2778,8 +2803,9 @@ export class TUI extends Container { // verified once (a pending header settling, a barrier clearing above a // shifted tail); rows past the boundary are still-live frozen snapshots, // exempt so a collapsing preview can never spray re-anchors mid-run. A - // divergence re-anchors and recommits — duplication, never loss — - // instead of silently skipping rows (committed nowhere, painted + // divergence re-anchors — feeding the divergenceRebuild erase-and-replay + // below (mux fallback: recommit below the stale copy; duplication, never + // loss) — instead of silently skipping rows (committed nowhere, painted // nowhere). Skipped on geometry frames (a rewrap legitimately reflows // every row), and skipped when the composed frame's stable prefix // covers every verified row and no rows newly became final. @@ -2852,7 +2878,21 @@ export class TUI extends Container { const firstPaint = !this.#hasEverRendered; const replaceRequested = this.#clearScrollbackOnNextRender; const geometryRebuild = geometryChanged && !resizeRepaintsInPlace(); - const fullPaint = firstPaint || replaceRequested || geometryRebuild; + // Committed history no longer matches the frame: a finalized block + // replaced its scrolled-off live render, or the frame collapsed into + // recorded rows. Native scrollback is a render cache, not a court + // record — erase and replay so history holds the content exactly once, + // instead of recommitting the final form below the stale fragment + // (a visibly duplicated block). Multiplexer panes cannot ED3 safely + // and keep the repair-below fallback in the branches under this one. + const divergenceRebuild = + this.#scrollbackRebuildEnabled && + !firstPaint && + !replaceRequested && + !geometryChanged && + !isMultiplexerSession() && + (committedRowsResynced || frameLength <= this.#committedRows); + const fullPaint = firstPaint || replaceRequested || geometryRebuild || divergenceRebuild; let windowTop: number; let chunkTo: number; if (fullPaint) { @@ -2865,16 +2905,17 @@ export class TUI extends Container { frameLength - this.#committedRows < height && cursorMarkers.some(marker => marker.row >= this.#committedRows)) ) { - // Either the frame shrank into the committed prefix, or a - // committed-prefix resync left a focused cursor tail shorter than the - // viewport. The latter happens when a streaming/live block had an - // append-only prefix committed, then collapses on abort/finalize: - // the audit re-anchors #committedRows at the first divergent row, but - // flooring windowTop there would pin the editor near the top and - // leave blank rows underneath. Re-show the frame tail instead. The - // stale committed copy stays in native history; duplicating a few rows - // is preferable to a live editor gap and matches the existing - // "duplication, never loss" resync contract. + // Multiplexer fallback (a direct terminal takes the divergenceRebuild + // full paint above): either the frame shrank into the committed + // prefix, or a committed-prefix resync left a focused cursor tail + // shorter than the viewport. The latter happens when a streaming/live + // block had an append-only prefix committed, then collapses on + // abort/finalize: the audit re-anchors #committedRows at the first + // divergent row, but flooring windowTop there would pin the editor + // near the top and leave blank rows underneath. Re-show the frame + // tail instead. The stale committed copy stays in native history; + // duplicating a few rows is preferable to a live editor gap — + // "duplication, never loss" is the ED3-unsafe fallback contract. committedPrefixResliced = true; windowTop = Math.max(0, frameLength - height); chunkTo = windowTop; @@ -2927,7 +2968,10 @@ export class TUI extends Container { const cursorTrackingLineCount = hasVisibleOverlay ? Math.max(frame.length, windowTop + height) : frame.length; const intent: RenderIntent = fullPaint - ? { kind: "fullPaint", clearScrollback: replaceRequested || geometryRebuild ? !isMultiplexerSession() : false } + ? { + kind: "fullPaint", + clearScrollback: divergenceRebuild || ((replaceRequested || geometryRebuild) && !isMultiplexerSession()), + } : { kind: "update", chunkTo, windowTop }; this.#logRedraw(intent, frameLength, height); diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 526c73d6d..3065f5df4 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -1,3 +1,5 @@ +process.env.PI_TUI_SCROLLBACK_REBUILD = "true"; + import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { type Component, @@ -1746,7 +1748,7 @@ describe("TUI terminal-state regressions", () => { tui.stop(); } }); - it("recommits an offscreen expansion behind the stale prefix while seam commits continue in order", async () => { + it("rebuilds history when an offscreen expansion lands with a tail append", async () => { const term = new VirtualTerminal(32, 6); const tui = new TUI(term); const component = new MutableLinesComponent(["status-0", ...rows("line-", 11)]); @@ -1767,7 +1769,8 @@ describe("TUI terminal-state regressions", () => { // Rows 0..5 (status-0, line-0..line-4) are committed. The frame edits // row 0 and inserts a row above the commit boundary while a tail // append lands in the same frame: 2+ prefix tail samples change, so - // the committed-prefix audit re-anchors at row 0 and recommits. + // the committed-prefix audit re-anchors and the engine erases and + // replays — history holds the expanded frame exactly once. component.setLines(["status-1", "expanded-details", ...rows("line-", 12)]); tui.requestRender(); await settle(term); @@ -1782,19 +1785,11 @@ describe("TUI terminal-state regressions", () => { ]); const buffer = term.getScrollBuffer().map(line => line.trimEnd()); const history = buffer.slice(0, term.getBufferPosition().baseY); - // RESYNC law: native history keeps the stale committed copy AND gains - // a fresh copy of the diverged frame from row 0 — the offscreen edit - // and the expansion reach history (duplication, never loss). - expect(history).toEqual([ - // stale committed prefix, never rewritten - "status-0", - ...rows("line-", 5), - // recommitted frame rows 0..7 (new committed = 14 - height) - "status-1", - "expanded-details", - ...rows("line-", 6), - ]); - // The appended tail row reaches the screen exactly once. + // REBUILD law: the stale committed copy is erased; history is the + // diverged frame's own prefix — the offscreen edit and the expansion + // reach history exactly once. + expect(history).toEqual(["status-1", "expanded-details", ...rows("line-", 6)]); + expect(buffer).not.toContain("status-0"); expect(buffer.filter(row => row === "line-11").length).toBe(1); } finally { tui.stop(); @@ -1836,11 +1831,11 @@ describe("TUI terminal-state regressions", () => { } }); - it("keeps stale collapsed ctrl-o markers in history and recommits the expanded rows behind them", async () => { + it("erases stale collapsed ctrl-o markers and rebuilds history with the expanded rows", async () => { // A Ctrl+O expansion mutates committed rows, so the committed-prefix - // audit resyncs: the collapsed markers that already scrolled into native - // history stay there — one stale copy each, never rewritten — and the - // expanded rows recommit behind them (duplication, never loss). + // audit resyncs and the engine erases-and-replays: the collapsed + // markers that already scrolled into native history are erased with + // the rebuild and the expanded rows land in history exactly once. const term = new VirtualTerminal(48, 6); const tui = new TUI(term); const collapsedLines = [ @@ -1871,7 +1866,6 @@ describe("TUI terminal-state regressions", () => { ]); tui.requestRender(); await settle(term); - const history = term.getScrollBuffer().slice(0, term.getBufferPosition().baseY); expect(visible(term).map(line => line.trim())).toEqual([ "json-6", @@ -1881,79 +1875,32 @@ describe("TUI terminal-state regressions", () => { "status", "editor", ]); - const scrollback = term.getScrollBuffer(); - expect(countMatches(scrollback, /Ctrl\+O: Expand/)).toBe(1); - expect(countMatches(scrollback, /ctrl\+o/)).toBe(1); - // The resync re-anchors at the first diverged row (the code marker) - // and recommits from there: the expanded rows reach history exactly - // once, right behind the stale markers. - for (const line of ["code line 0", "code line 1", "output line 0", "output line 1"]) { - expect(countMatches(history, new RegExp(`^${line}\\s*$`)), `${line} recommits exactly once`).toBe(1); - } - // json rows inside the recommitted span carry one stale + one fresh - // copy; rows still in the live window appear exactly once. - for (let i = 0; i < 6; i++) { - const pattern = new RegExp(`\\bjson-${i}\\b`); - expect(countMatches(scrollback, pattern), `json-${i} appears twice (stale + recommit)`).toBe(2); - } - for (let i = 6; i < 10; i++) { - const pattern = new RegExp(`\\bjson-${i}\\b`); - expect(countMatches(scrollback, pattern), `json-${i} should appear exactly once`).toBe(1); - } - } finally { - tui.stop(); - } - }); - - it("defers offscreen expansion rebuild when the viewport position is unknown", async () => { - // POSIX terminals cannot report whether the user scrolled up, so an - // ordinary offscreen expansion must NOT destructively rebuild scrollback - // (anti-yank). The collapsed ctrl+o markers that scrolled into history - // therefore stay stale until the next checkpoint — this is the deferral - // that makes an un-flagged Ctrl+O expand look broken above the fold. - const term = new UnknownViewportTerminal(48, 6); - const tui = new TUI(term); - const component = new MutableLinesComponent([ - "frame-top", - "code preview … 16 more lines ⟨Ctrl+O: Expand⟩", - "output preview … 106 more lines (ctrl+o to expand)", - ...rows("json-", 10), - "status", - "editor", - ]); - tui.addChild(component); - - try { - tui.start(); - await settle(term); - expect(term.isNativeViewportAtBottom()).toBeUndefined(); - expect(term.getScrollBuffer().join("\n")).toContain("ctrl+o"); - - component.setLines([ + // Rebuilt history: the expanded rows exactly once, no stale markers + // anywhere on the tape. + expect(term.getScrollBuffer().join("\n")).not.toContain("ctrl+o"); + expect(term.getScrollBuffer().join("\n")).not.toContain("Ctrl+O"); + const history = term + .getScrollBuffer() + .slice(0, term.getBufferPosition().baseY) + .map(line => line.trimEnd()); + expect(history).toEqual([ "frame-top", "code line 0", "code line 1", "output line 0", "output line 1", - ...rows("json-", 10), - "status", - "editor", + ...rows("json-", 6), ]); - tui.requestRender(); - await settle(term); - - // No flag: the rebuild is deferred, so the stale markers survive offscreen. - expect(term.getScrollBuffer().join("\n")).toContain("ctrl+o"); } finally { tui.stop(); } }); - it("paints an offscreen expansion identically when the viewport probe is unavailable", async () => { - // Law 3: there is no probe and no platform fork. A terminal that cannot - // report its native viewport position gets exactly the same treatment as - // one that can: committed rows stay immutable, the live window repaints, - // and no clear/home bytes are emitted. + it("rebuilds an offscreen expansion identically when the viewport probe is unavailable", async () => { + // There is no probe and no platform fork: a terminal that cannot + // report its native viewport position gets exactly the same + // erase-and-replay as one that can — exactly-once history wins over + // the parked-reader anchor (upstream pi semantics). const term = new UnknownViewportTerminal(48, 6); const tui = new TUI(term); const component = new MutableLinesComponent([ @@ -1986,9 +1933,7 @@ describe("TUI terminal-state regressions", () => { tui.requestRender(); await settle(term); - const paint = writes.join(""); - expect(paint).not.toContain("\x1b[3J"); - expect(paint).not.toContain("\x1b[2J"); + expect((writes.join("").match(/\x1b\[3J/g) ?? []).length).toBe(1); expect(visible(term).map(line => line.trim())).toEqual([ "json-6", "json-7", @@ -1997,8 +1942,8 @@ describe("TUI terminal-state regressions", () => { "status", "editor", ]); - // History is never rewritten: the stale markers survive offscreen. - expect(term.getScrollBuffer().join("\n")).toContain("ctrl+o"); + // History was rebuilt, not left stale: the markers are gone. + expect(term.getScrollBuffer().join("\n")).not.toContain("ctrl+o"); } finally { tui.stop(); } @@ -2053,14 +1998,11 @@ describe("TUI terminal-state regressions", () => { } }); - it("re-anchors a bottom-anchored high-water collapse at the divergence and recommits the tail into history", async () => { - // RESYNC law: the collapse shrinks the frame below the committed count, - // so the engine re-anchors the commit index at the first diverged row - // (row 8, where preview-* became result-*). Native history is - // append-only: the high-water preview copy stays in scrollback above - // (accepted artifact) — never clawed back. The window starts at the - // re-anchored commit index, which here sits past `length - height`, so - // the short tail is blank-padded rather than overwriting committed rows. + it("rebuilds a bottom-anchored high-water collapse with the final tail exactly once", async () => { + // REBUILD law: the collapse shrinks the frame below the committed + // count, so the engine erases and replays the final frame — the + // high-water preview copy is erased instead of surviving above as a + // stale artifact, and the result rows are history's only copy. const term = new VirtualTerminal(40, 5); const highWaterFrame = [...rows("base-", 8), ...rows("preview-", 10)]; const finalFrame = [...rows("base-", 8), "result-0", "result-1"]; @@ -2079,13 +2021,18 @@ describe("TUI terminal-state regressions", () => { tui.requestRender(); await settle(term); - expect(writes.join("")).not.toContain("\x1b[3J"); - expect(visible(term).map(line => line.trim())).toEqual(["result-0", "result-1", "", "", ""]); - const history = term.getScrollBuffer().slice(0, term.getBufferPosition().baseY); - expect(history.map(line => line.trimEnd())).toEqual(highWaterFrame.slice(0, 13)); + expect((writes.join("").match(/\x1b\[3J/g) ?? []).length).toBe(1); + expect(term.getScrollBuffer().map(line => line.trimEnd())).toEqual(finalFrame); + expect(visible(term).map(line => line.trim())).toEqual([ + "base-5", + "base-6", + "base-7", + "result-0", + "result-1", + ]); // Once the transcript grows past the window again, the post-collapse - // tail commits: result rows REACH history instead of being ignored. + // tail commits normally on the update path — exactly once. component.setLines([...finalFrame, ...rows("tail-", 5)]); tui.requestRender(); await settle(term); @@ -2095,7 +2042,7 @@ describe("TUI terminal-state regressions", () => { .getScrollBuffer() .slice(0, term.getBufferPosition().baseY) .map(line => line.trimEnd()); - expect(grownHistory).toEqual([...highWaterFrame.slice(0, 13), "result-0", "result-1"]); + expect(grownHistory).toEqual(finalFrame); } finally { tui.stop(); } @@ -2127,12 +2074,11 @@ describe("TUI terminal-state regressions", () => { } }); - it("recommits an offscreen expansion at the seam while the reader is parked in scrollback", async () => { + it("rebuilds an offscreen expansion and snaps a parked reader to the tail", async () => { // An expansion above the commit boundary triggers the committed-prefix - // resync: the inserted rows (and the shifted committed rows) recommit - // at the seam, so the expansion reaches native history instead of - // being skipped. The reader scrolled into scrollback keeps a stable - // view — the recommit only appends below their anchor. + // resync: the engine erases and replays, so the expansion reaches + // native history exactly once. The parked reader is snapped to the + // bottom — the price of exactly-once history (upstream pi semantics). const term = new VirtualTerminal(32, 5); const tui = new TUI(term); const component = new MutableLinesComponent(rows("line-", 12)); @@ -2151,23 +2097,13 @@ describe("TUI terminal-state regressions", () => { tui.requestRender(); await settle(term); - const paint = writes.join(""); - expect(paint).not.toContain("\x1b[3J"); - expect(paint).not.toContain("\x1b[2J"); - expect(paint).not.toContain("\x1b[H"); - expect(term.getBufferPosition().viewportY).toBe(before.viewportY); - expect(visible(term).map(line => line.trim())).toEqual(["line-2", "line-3", "line-4", "line-5", "line-6"]); - // The resync recommits the expansion: it reaches native history - // exactly because the commit index re-anchored at the divergence. - expect(term.getScrollBuffer().join("\n")).toContain("expanded-0"); + expect((writes.join("").match(/\x1b\[3J/g) ?? []).length).toBe(1); + const buffer = term.getScrollBuffer().map(line => line.trimEnd()); + expect(buffer).toEqual(["line-0", "line-1", "expanded-0", "expanded-1", ...rows("line-", 12).slice(2)]); term.scrollLines(999); tui.requestRender(); await settle(term); - - const finalPosition = term.getBufferPosition(); - expect(finalPosition.viewportY).toBe(finalPosition.baseY); - expect(term.getScrollBuffer().join("\n")).toContain("expanded-0"); expect(visible(term).map(line => line.trim())).toEqual([ "line-7", "line-8", @@ -2356,13 +2292,12 @@ describe("TUI terminal-state regressions", () => { scrolledTui.stop(); } }); - it("re-anchors a huge completion-style collapse at the new tail and recommits the diverged head behind stale history", async () => { - // RESYNC law (#1599 lineage): a 100-row transcript collapsing to 20 rows - // in a 10-row window must keep the new tail (including the prompt) on - // screen. The frame no longer covers the committed prefix, so the commit - // index re-anchors at the first diverged row (row 0) and recommits up to - // `newLength - height`: short-0..short-9 land in history right behind - // the stale line-* copy — duplication of the stale prefix, never loss. + it("rebuilds a huge completion-style collapse with the new tail exactly once", async () => { + // REBUILD law (#1599 lineage): a 100-row transcript collapsing to 20 + // rows in a 10-row window keeps the new tail (including the prompt) on + // screen. The frame no longer covers the committed prefix, so the + // engine erases and replays: short-0..short-9 are history's only + // rows — the stale line-* transcript is gone. const term = new UnknownViewportTerminal(40, 10); const tui = new TUI(term); const body = rows("line-", 99); @@ -2391,25 +2326,20 @@ describe("TUI terminal-state regressions", () => { "short-18", "prompt-row", ]); - const buffer = term.getScrollBuffer(); - // The recommit puts short-0..short-9 into history exactly once and - // the re-anchored window holds short-10..prompt-row exactly once — - // nothing is lost, nothing duplicates. - for (let i = 0; i < short.length; i++) { - expect(countMatches(buffer, new RegExp(`\\bshort-${i}\\b`)), `short-${i} appears once`).toBe(1); - } - const history = buffer.slice(0, term.getBufferPosition().baseY).map(line => line.trimEnd()); - expect(history).toEqual([...body.slice(0, 90), ...rows("short-", 10)]); + const buffer = term.getScrollBuffer().map(line => line.trimEnd()); + expect(buffer).toEqual([...short, "prompt-row"]); + const history = buffer.slice(0, term.getBufferPosition().baseY); + expect(history).toEqual(rows("short-", 10)); } finally { tui.stop(); } }); - it("recommits a huge collapse without clears while the reader is parked in scrollback", async () => { - // RESYNC + no-clear law: even a 100→20 row collapse never emits ED2/ED3 - // — the re-anchor recommits the diverged frame head behind the stale - // prefix and rewrites the window, so a reader parked in native - // scrollback keeps a byte-stable view and their anchor. + it("rebuilds a huge collapse and snaps a parked reader to the tail", async () => { + // REBUILD law: a 100→20 row collapse erases and replays (one ED3). + // The reader parked in native scrollback is snapped to the bottom — + // stale history is no longer preserved for them (upstream pi + // semantics: exactly-once history wins over the anchored view). const term = new UnknownViewportTerminal(40, 10); const tui = new TUI(term); const body = rows("line-", 99); @@ -2420,9 +2350,7 @@ describe("TUI terminal-state regressions", () => { tui.start(); await settle(term); term.scrollLines(-10); - const before = term.getBufferPosition(); - const beforeViewport = visible(term).map(line => line.trim()); - expect(before.viewportY).toBeGreaterThan(0); + expect(term.getBufferPosition().viewportY).toBeGreaterThan(0); const writes = captureWrites(term); const short = rows("short-", 19); @@ -2430,11 +2358,7 @@ describe("TUI terminal-state regressions", () => { tui.requestRender(); await settle(term); - const paint = writes.join(""); - expect(paint).not.toContain("\x1b[3J"); - expect(paint).not.toContain("\x1b[2J"); - expect(term.getBufferPosition().viewportY).toBe(before.viewportY); - expect(visible(term).map(line => line.trim())).toEqual(beforeViewport); + expect((writes.join("").match(/\x1b\[3J/g) ?? []).length).toBe(1); term.scrollLines(999); await settle(term); @@ -2451,21 +2375,18 @@ describe("TUI terminal-state regressions", () => { "short-18", "prompt-row", ]); - // Stale committed history stays above — never clawed back — and the - // recommitted frame head follows it (no row loss). - const history = term.getScrollBuffer().slice(0, term.getBufferPosition().baseY); - expect(history.join("\n")).toContain("line-89"); - for (let i = 0; i < 10; i++) { - expect(countMatches(history, new RegExp(`\\bshort-${i}\\b`)), `short-${i} recommits once`).toBe(1); - } - for (let i = 10; i < short.length; i++) { - expect(countMatches(history, new RegExp(`\\bshort-${i}\\b`)), `short-${i} stays in the window`).toBe(0); - } + // History was rebuilt: the short head exactly once, the stale + // line-* transcript erased. + const history = term + .getScrollBuffer() + .slice(0, term.getBufferPosition().baseY) + .map(line => line.trimEnd()); + expect(history).toEqual(rows("short-", 10)); } finally { tui.stop(); } }); - it("resyncs an offscreen-edit grow and re-anchors the following collapse", async () => { + it("rebuilds on an offscreen-edit grow and again on the following collapse", async () => { const term = new UnknownViewportTerminal(40, 10); const tui = new TUI(term); const initial = rows("line-", 19); @@ -2477,9 +2398,9 @@ describe("TUI terminal-state regressions", () => { await settle(term); // Offscreen edit (row 0) + 100-row growth in one frame: the edit is - // an insertion above the commit boundary, so the audit re-anchors and - // the edited transcript recommits behind the stale original (law 1 - // content-at-commit-time, duplication never loss). + // an insertion above the commit boundary, so the audit re-anchors + // and the engine erases-and-replays — the edited transcript is + // history's only copy. const expanded = ["edited-line", ...rows("line-", 118), "prompt-row"]; component.setLines(expanded); tui.requestRender(); @@ -2496,10 +2417,14 @@ describe("TUI terminal-state regressions", () => { "line-117", "prompt-row", ]); - expect(term.getScrollBuffer().join("\n")).toContain("edited-line"); + const grownHistory = term + .getScrollBuffer() + .slice(0, term.getBufferPosition().baseY) + .map(line => line.trimEnd()); + expect(grownHistory).toEqual(expanded.slice(0, 110)); - // Collapse far below the commit boundary: law 4 re-anchors the window - // at the new tail; the stale committed transcript stays above. + // Collapse far below the commit boundary: a second rebuild replays + // the short frame; the tall transcript is erased. const short = [...rows("short-", 14), "prompt-row"]; component.setLines(short); tui.requestRender(); @@ -2517,10 +2442,11 @@ describe("TUI terminal-state regressions", () => { "short-13", "prompt-row", ]); - const history = term.getScrollBuffer().slice(0, term.getBufferPosition().baseY).join("\n"); - expect(history).toContain("line-108"); - expect(history).toContain("short-4"); - expect(history).toContain("edited-line"); + const history = term + .getScrollBuffer() + .slice(0, term.getBufferPosition().baseY) + .map(line => line.trimEnd()); + expect(history).toEqual(rows("short-", 5)); } finally { tui.stop(); } diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts index 7098aceaf..945217ecd 100644 --- a/packages/tui/test/streaming-scrollback-defer.test.ts +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -1,3 +1,5 @@ +process.env.PI_TUI_SCROLLBACK_REBUILD = "true"; + import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import { type Component, @@ -10,7 +12,7 @@ import { VirtualTerminal } from "./virtual-terminal"; // Law-encoding suite for native-scrollback commits. // // The tape is the terminal's visual record: whatever scrolls above the window -// enters history exactly once, in order. The component seam +// enters history, in order. The component seam // (`getNativeScrollbackLiveRegionStart`) classifies HOW a row commits: // ► below the boundary — exact-final bytes, hard-verified, audited; // ► above the boundary — a frozen snapshot of what was on screen, exempt @@ -18,9 +20,11 @@ import { VirtualTerminal } from "./virtual-terminal"; // never spray duplicates mid-run); // ► when the boundary rises past frozen snapshots (the block finalized, a // barrier cleared), they are strict-scanned exactly once: a divergence -// re-anchors and recommits the final content below the frozen snapshot — -// duplication, never loss; rows are never committed-nowhere-and-painted- -// nowhere. +// erases native history and replays the frame (one ED3), so the tape +// holds the final content exactly once — never a stale fragment above a +// recommit. Multiplexer panes, where ED3 is unsafe, keep the repair-below +// fallback: the final content recommits below the frozen snapshot — +// duplication, never loss. class LineList implements Component { #lines: string[]; @@ -306,7 +310,7 @@ describe("streaming scrollback — visual record", () => { } }); - it("repairs a wholesale-replaced live block once at finalize — full result, single stale fragment", async () => { + it("rebuilds history once at finalize when a wholesale-replaced live block diverged", async () => { if (process.platform === "win32") return; const term = new VirtualTerminal(24, 4); overrideProbe(term, undefined); @@ -342,16 +346,15 @@ describe("streaming scrollback — visual record", () => { ]); // Finalize: the one-time strict verification catches the divergence - // and recommits the final content below the frozen fragment. Every - // fresh row is on the tape (no loss); the stale fragment appears - // exactly once (no spray). + // and erases-and-replays, so the tape holds the final content exactly + // once — the stale frozen fragment is gone, nothing recommits below it. live.seam = undefined; tui.requestRender(); await settle(term); const buffer = tape(term); - expect(eraseScrollbackCount(writes)).toBe(0); - expect(buffer).toEqual([...rows("prior-", 12), ...rows("pending-stale-", 6), ...rows("running-fresh-", 10)]); + expect(eraseScrollbackCount(writes)).toBe(1); + expect(buffer).toEqual([...rows("prior-", 12), ...rows("running-fresh-", 10)]); } finally { tui.stop(); } @@ -387,16 +390,21 @@ describe("streaming scrollback — visual record", () => { tui.requestRender(); await settle(term); + // Mid-run: frozen snapshots are exempt — a wholesale replace while + // live must not trigger a rebuild. A lower sibling's seam winning + // would verify the live rows as final and re-anchor right here. + expect(eraseScrollbackCount(writes)).toBe(0); + live.seam = undefined; tui.requestRender(); await settle(term); const buffer = tape(term); - expect(eraseScrollbackCount(writes)).toBe(0); - // Full fresh content present in order (no loss), stale head fragment - // exactly once (no spray), loader still live at the bottom. + // Finalize rebuild: full fresh content exactly once, the stale + // preview fragment erased, loader still live at the bottom. + expect(eraseScrollbackCount(writes)).toBe(1); expect(contiguousAt(buffer, rows("running-fresh-", 10))).toHaveLength(1); - expect(buffer.filter(line => line.startsWith("pending-stale-"))).toEqual(rows("pending-stale-", 7)); + expect(buffer.filter(line => line.startsWith("pending-stale-"))).toEqual([]); expect(buffer.at(-1)).toBe("Working..."); } finally { tui.stop(); @@ -502,14 +510,14 @@ describe("streaming scrollback — visual record", () => { await settle(term); expect(term.getScrollBuffer().filter(line => line.startsWith("prior-"))).toEqual(rows("prior-", 12)); - // Live block collapses to its compact result. The bottom-anchored - // viewport would re-expose committed sealed rows; the pin must clamp - // the repaint to the committed boundary instead of duplicating them. + // Live block collapses to its compact result: the frame shrank into + // recorded rows, so history is rebuilt — the sealed rows appear + // exactly once, never appended a second time below their old copy. live.setLines(["done"]); tui.requestRender(); await settle(term); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); expect(term.getScrollBuffer().filter(line => line.startsWith("prior-"))).toEqual(rows("prior-", 12)); } finally { tui.stop(); @@ -536,16 +544,17 @@ describe("streaming scrollback — visual record", () => { expect(eraseScrollbackCount(writes)).toBe(0); - // A later frame introduces a live region after the same sealed prefix. - // The already-committed base rows must stay accounted — never appended - // to native history a second time. + // A later frame introduces a live region after the same sealed prefix + // and drops the transient tail: the frame shrank into recorded rows, + // so one rebuild replays history with the base rows exactly once — + // never appended to native history a second time. const live = new SeamLineList(rows("live-", 20)); sealed.setLines(rows("base-", 12)); tui.addChild(live); tui.requestRender(); await settle(term); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); expect(term.getScrollBuffer().filter(line => line.startsWith("base-"))).toEqual(rows("base-", 12)); } finally { tui.stop(); @@ -745,7 +754,7 @@ describe("streaming scrollback — visual record", () => { } }); - it("never re-anchors a re-laying-out live block mid-run, repairs once at finalize", async () => { + it("never re-anchors a re-laying-out live block mid-run, rebuilds once at finalize", async () => { if (process.platform === "win32") return; const term = new VirtualTerminal(20, 4); overrideProbe(term, undefined); @@ -753,7 +762,7 @@ describe("streaming scrollback — visual record", () => { // A block that rewrites an interior row every frame (a streaming table // re-aligning, a collapsing preview). Its scrolled rows are frozen // snapshots: drift never sprays re-anchors; the single strict scan at - // finalize recommits the final form once. + // finalize erases-and-replays the final form once. const live = new SeamLineList([]); try { @@ -772,8 +781,9 @@ describe("streaming scrollback — visual record", () => { } // Mid-run: exactly the scrolled snapshots + the grid — one copy each, - // no spray despite nine drift frames. + // no spray and no rebuild despite nine drift frames. const streaming = tape(term); + expect(eraseScrollbackCount(writes)).toBe(0); expect(streaming).toHaveLength(12); expect(streaming.filter(line => line.startsWith("tbl-1 ")).length).toBe(1); @@ -781,32 +791,34 @@ describe("streaming scrollback — visual record", () => { tui.requestRender(); await settle(term); - // Finalize: one repair recommits the final layout below the frozen - // snapshot; the final form of the drifted row is on the tape. + // Finalize: one erase-and-replay puts the final layout on the tape + // exactly once — the drifted row's final form is the only copy. + const finalLines = rows("tbl-", 12); + finalLines[1] = "tbl-1 [w12]"; const buffer = tape(term); - expect(eraseScrollbackCount(writes)).toBe(0); - expect(buffer.join("\n")).toContain("tbl-1 [w12]"); - // Bounded: 8 snapshots + one repair recommit (7 rows) + 4 grid rows. - expect(buffer.length).toBeLessThanOrEqual(19); + expect(eraseScrollbackCount(writes)).toBe(1); + expect(buffer).toEqual(finalLines); - // Stability: identical follow-up frames must not grow the tape. + // Stability: identical follow-up frames must not grow the tape or + // erase again. tui.requestRender(); await settle(term); expect(tape(term)).toEqual(buffer); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } }); - it("repairs a declared-final violation by re-anchoring once, never spraying", async () => { + it("repairs a declared-final violation with one rebuild, never spraying", async () => { if (process.platform === "win32") return; const term = new VirtualTerminal(20, 4); overrideProbe(term, undefined); const tui = new TUI(term); // The block declares its whole body final, commits, then violates the // contract by rewriting TWO committed rows (alignment breaks, so the - // tail-sample tolerance cannot absorb it). The audit re-anchors and - // recommits — duplication, never loss — and stays quiet afterwards. + // tail-sample tolerance cannot absorb it). The audit re-anchors, the + // engine erases-and-replays once, and stays quiet afterwards. const live = new SeamLineList(rows("row-", 12)); live.seam = Number.POSITIVE_INFINITY; @@ -824,15 +836,15 @@ describe("streaming scrollback — visual record", () => { await settle(term); const afterViolation = tape(term); - expect(afterViolation).toContain("row-5 [edited]"); - expect(afterViolation).toContain("row-6 [edited]"); + expect(eraseScrollbackCount(writes)).toBe(1); + expect(afterViolation).toEqual(violated); for (let i = 0; i < 5; i++) { tui.requestRender(); await settle(term); } expect(tape(term)).toEqual(afterViolation); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -871,18 +883,17 @@ describe("scrollback commit gap — live barriers", () => { expect(tape(term)).toEqual(["[tool pending]", ...rows("ans-", 8)]); // Barrier removed: the tail shifts up. The one-time strict scan - // catches the shift and recommits — every ans row survives, in order, - // contiguous at the tape bottom. + // catches the shift and rebuilds — every ans row survives, in order, + // and the stale barrier row is erased from history. root.setLines(rows("ans-", 8)); root.seam = undefined; tui.requestRender(); await settle(term); const buffer = tape(term); - expect(buffer.slice(-8)).toEqual(rows("ans-", 8)); - expect(buffer.filter(line => line === "[tool pending]")).toHaveLength(1); + expect(buffer).toEqual(rows("ans-", 8)); expect(term.getViewport().map(line => line.trimEnd())).toEqual(rows("ans-", 8).slice(-4)); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -915,12 +926,11 @@ describe("scrollback commit gap — live barriers", () => { await settle(term); const buffer = tape(term); - // Full result contiguous at the bottom; the recorded preview head - // stays above it as the visual record — once, no spray. - expect(buffer.slice(-9)).toEqual(result); - expect(contiguousAt(buffer, result)).toHaveLength(1); + // History rebuilt: the full result exactly once; the provisional + // preview is erased rather than left above as a stale record. + expect(buffer).toEqual(result); expect(term.getViewport().map(line => line.trimEnd())).toEqual(result.slice(-4)); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -948,7 +958,7 @@ describe("scrollback commit gap — live barriers", () => { // Barrier collapses to 1 row but the frame stays longer than the // committed prefix (NOT the shrink-into-prefix branch); the strict - // scan must catch the upward tail shift. + // scan must catch the upward tail shift and rebuild. const f2 = ["bar-collapsed", ...rows("tail-", 8)]; root.setLines(f2); root.seam = undefined; @@ -956,9 +966,9 @@ describe("scrollback commit gap — live barriers", () => { await settle(term); const buffer = tape(term); - expect(buffer.slice(-9)).toEqual(f2); + expect(buffer).toEqual(f2); expect(term.getViewport().map(line => line.trimEnd())).toEqual(f2.slice(-4)); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -986,16 +996,16 @@ describe("scrollback commit gap — live barriers", () => { expect(tape(term)).toEqual(["[tool pending]", ...rows("out-", 10)]); // Remove the barrier. The tail shifts up by one row; the strict scan - // recommits so every out-* row remains, in order, contiguous at the - // tape bottom. + // rebuilds so every out-* row remains, in order, and the stale + // barrier row is erased. tui.removeChild(barrier); tui.requestRender(); await settle(term); const buffer = tape(term); - expect(buffer.slice(-10)).toEqual(rows("out-", 10)); + expect(buffer).toEqual(rows("out-", 10)); expect(term.getViewport().map(line => line.trimEnd())).toEqual(rows("out-", 10).slice(-5)); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -1030,8 +1040,7 @@ describe("scrollback commit gap — live barriers", () => { await settle(term); const buffer = tape(term); - expect(buffer.slice(-20)).toEqual(final); - expect(buffer.filter(line => line === "[pending]")).toHaveLength(1); + expect(buffer).toEqual(final); expect(term.getViewport().map(line => line.trimEnd())).toEqual(final.slice(-5)); } finally { tui.stop(); @@ -1104,7 +1113,7 @@ describe("scrollback commit gap — live barriers", () => { // Finalize: ONLY row 0 changes (preview → result); the whole tail is // byte-identical. The tail-sample tolerance alone would eat the single // mismatch and "result" would never reach the tape; the strict scan of - // the newly-final span forces the recommit. + // the newly-final span forces the rebuild. const f2 = ["result", ...rows("tail-", 8)]; root.setLines(f2); root.seam = undefined; @@ -1112,10 +1121,10 @@ describe("scrollback commit gap — live barriers", () => { await settle(term); const buffer = tape(term); - expect(buffer).toContain("result"); + expect(buffer).toEqual(f2); expect(buffer.filter(line => line === "result")).toHaveLength(1); expect(term.getViewport().map(line => line.trimEnd())).toEqual(f2.slice(-4)); - expect(eraseScrollbackCount(writes)).toBe(0); + expect(eraseScrollbackCount(writes)).toBe(1); } finally { tui.stop(); } @@ -1148,8 +1157,63 @@ describe("scrollback commit gap — live barriers", () => { await settle(term); const buffer = tape(term); - expect(buffer).toContain("result"); + expect(buffer).toEqual(["result", ...rows("tail-", 30)]); expect(buffer.filter(line => line === "result")).toHaveLength(1); + expect(eraseScrollbackCount(writes)).toBe(1); + } finally { + tui.stop(); + } + }); +}); + +describe("scrollback divergence — multiplexer fallback", () => { + let savedTerminalEnv: Record = {}; + let savedTmux: string | undefined; + beforeEach(() => { + savedTerminalEnv = saveTerminalEnv(); + savedTmux = Bun.env.TMUX; + Bun.env.TMUX = "/tmp/tmux-1000/default,12345,0"; + }); + afterEach(() => { + if (savedTmux === undefined) delete Bun.env.TMUX; + else Bun.env.TMUX = savedTmux; + restoreTerminalEnv(savedTerminalEnv); + savedTerminalEnv = {}; + }); + + it("repairs below the stale fragment without ED3 when the pane cannot be cleared", async () => { + if (process.platform === "win32") return; + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + const tui = new TUI(term); + const root = new SeamLineList([]); + + try { + tui.addChild(root); + tui.start(); + await settle(term); + const writes = capture(term); + + const preview = rows("preview-", 10); + root.setLines(preview); + root.seam = 0; + tui.requestRender(); + await settle(term); + expect(tape(term)).toEqual(preview); + + // Finalize divergence inside a tmux pane: ED3 would corrupt the + // pane's own history, so the engine keeps the repair-below contract — + // the full result reaches the tape contiguously, the frozen preview + // head stays above it exactly once, and nothing is erased. + const result = rows("result-", 9); + root.setLines(result); + root.seam = undefined; + tui.requestRender(); + await settle(term); + + const buffer = tape(term); + expect(buffer.slice(-9)).toEqual(result); + expect(contiguousAt(buffer, result)).toHaveLength(1); expect(eraseScrollbackCount(writes)).toBe(0); } finally { tui.stop(); From 77f641268d9802a686453435d6d67e381a269380 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 14 Jul 2026 01:30:40 +0200 Subject: [PATCH 24/28] feat(metaharness): migrated harbor-manager to metaharness and updated harness logic - Renamed and migrated package-level artifacts from `harbor-manager` to `metaharness`. - Updated metaharness core services (`server`, `runner`, `store`, `experiments`) with substantive logic edits. - Synced CLI, local agent, and config wiring to align with the metaharness package structure. - Refreshed benchmark-related documentation/prompts and expanded tests for the revised metaharness flow. --- bun.lock | 32 ++-- package.json | 2 +- .../{harbor-manager => metaharness}/README.md | 24 ++- .../adapters/edit/bun-imports.d.ts | 0 .../adapters/edit/cli.ts | 2 +- .../adapters/edit/prompts/benchmark-retry.md | 0 .../adapters/edit/prompts/benchmark-system.md | 0 .../adapters/edit/prompts/benchmark-task.md | 0 .../adapters/edit/report.ts | 0 .../adapters/edit/runner.test.ts | 0 .../adapters/edit/runner.ts | 0 .../adapters/edit/tsconfig.json | 0 .../agent/omp_local.py | 4 +- .../package.json | 6 +- .../src/adapters/snapcompact.py | 0 .../src/benchmarks.test.ts | 0 .../src/benchmarks.ts | 0 .../src/experiments.test.ts | 0 .../src/experiments.ts | 30 +++- .../src/launch-args.ts | 0 .../src/manager.test.ts | 101 ++++++++++- .../src/runner.test.ts | 0 .../src/runner.ts | 12 +- .../src/server.ts | 168 +++++++++++++++--- .../src/store.ts | 45 ++++- .../src/web/app.tsx | 6 +- .../src/web/index.html | 2 +- .../tsconfig.json | 0 28 files changed, 360 insertions(+), 74 deletions(-) rename packages/{harbor-manager => metaharness}/README.md (84%) rename packages/{harbor-manager => metaharness}/adapters/edit/bun-imports.d.ts (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/cli.ts (98%) rename packages/{harbor-manager => metaharness}/adapters/edit/prompts/benchmark-retry.md (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/prompts/benchmark-system.md (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/prompts/benchmark-task.md (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/report.ts (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/runner.test.ts (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/runner.ts (100%) rename packages/{harbor-manager => metaharness}/adapters/edit/tsconfig.json (100%) rename packages/{harbor-manager => metaharness}/agent/omp_local.py (99%) rename packages/{harbor-manager => metaharness}/package.json (92%) rename packages/{harbor-manager => metaharness}/src/adapters/snapcompact.py (100%) rename packages/{harbor-manager => metaharness}/src/benchmarks.test.ts (100%) rename packages/{harbor-manager => metaharness}/src/benchmarks.ts (100%) rename packages/{harbor-manager => metaharness}/src/experiments.test.ts (100%) rename packages/{harbor-manager => metaharness}/src/experiments.ts (94%) rename packages/{harbor-manager => metaharness}/src/launch-args.ts (100%) rename packages/{harbor-manager => metaharness}/src/manager.test.ts (80%) rename packages/{harbor-manager => metaharness}/src/runner.test.ts (100%) rename packages/{harbor-manager => metaharness}/src/runner.ts (99%) rename packages/{harbor-manager => metaharness}/src/server.ts (79%) rename packages/{harbor-manager => metaharness}/src/store.ts (92%) rename packages/{harbor-manager => metaharness}/src/web/app.tsx (99%) rename packages/{harbor-manager => metaharness}/src/web/index.html (93%) rename packages/{harbor-manager => metaharness}/tsconfig.json (100%) diff --git a/bun.lock b/bun.lock index 6e26f48e8..36dffdeec 100644 --- a/bun.lock +++ b/bun.lock @@ -136,11 +136,22 @@ "@types/react-dom": "catalog:", }, }, - "packages/harbor-manager": { - "name": "@oh-my-pi/harbor-manager", + "packages/hashline": { + "name": "@oh-my-pi/hashline", + "version": "16.4.8", + "dependencies": { + "diff": "catalog:", + "lru-cache": "catalog:", + }, + "devDependencies": { + "@types/bun": "catalog:", + }, + }, + "packages/metaharness": { + "name": "@oh-my-pi/pi-metaharness", "version": "0.0.1", "bin": { - "harbor-manager": "src/server.ts", + "metaharness": "src/server.ts", }, "dependencies": { "@oh-my-pi/hashline": "catalog:", @@ -167,17 +178,6 @@ "react-refresh": "^0.18.0", }, }, - "packages/hashline": { - "name": "@oh-my-pi/hashline", - "version": "16.4.8", - "dependencies": { - "diff": "catalog:", - "lru-cache": "catalog:", - }, - "devDependencies": { - "@types/bun": "catalog:", - }, - }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", "version": "16.4.8", @@ -759,8 +759,6 @@ "@oh-my-pi/collab-web": ["@oh-my-pi/collab-web@workspace:packages/collab-web"], - "@oh-my-pi/harbor-manager": ["@oh-my-pi/harbor-manager@workspace:packages/harbor-manager"], - "@oh-my-pi/hashline": ["@oh-my-pi/hashline@workspace:packages/hashline"], "@oh-my-pi/omp-stats": ["@oh-my-pi/omp-stats@workspace:packages/stats"], @@ -773,6 +771,8 @@ "@oh-my-pi/pi-coding-agent": ["@oh-my-pi/pi-coding-agent@workspace:packages/coding-agent"], + "@oh-my-pi/pi-metaharness": ["@oh-my-pi/pi-metaharness@workspace:packages/metaharness"], + "@oh-my-pi/pi-mnemopi": ["@oh-my-pi/pi-mnemopi@workspace:packages/mnemopi"], "@oh-my-pi/pi-natives": ["@oh-my-pi/pi-natives@workspace:packages/natives"], diff --git a/package.json b/package.json index da161644f..3005d3142 100644 --- a/package.json +++ b/package.json @@ -106,7 +106,7 @@ "collab:relay": "bun --cwd=packages/collab-web run relay", "collab:mock-host": "bun --cwd=packages/collab-web run mock-host", "collab:web:build": "bun --cwd=packages/collab-web run build", - "hmgr": "bun --cwd=packages/harbor-manager run dev", + "meta": "bun --cwd=packages/metaharness run dev", "claude:trace": "bun scripts/claude-trace.ts", "build": "bun run --workspaces --if-present build", "build:native": "bun --cwd=packages/natives run build", diff --git a/packages/harbor-manager/README.md b/packages/metaharness/README.md similarity index 84% rename from packages/harbor-manager/README.md rename to packages/metaharness/README.md index 85d614ca2..57330fe4c 100644 --- a/packages/harbor-manager/README.md +++ b/packages/metaharness/README.md @@ -1,4 +1,4 @@ -# @oh-my-pi/harbor-manager +# @oh-my-pi/pi-metaharness One manager for repository benchmarks. Harbor, TypeScript edit, and SnapCompact runs use the same experiment → run → trace model, SQLite store, REST/SSE API, @@ -30,8 +30,19 @@ bun run serve --port 4700 ## Server - `GET /` — experiments, runs, normalized traces, and a launch form for every benchmark. -- `GET /api/experiments` — experiment summaries across all benchmark types. -- `GET /api/runs` — uniform run rows with benchmark, score, progress, spend, and tokens. +- `GET /api/experiments[?q=]` — experiment summaries across all benchmark types + (`q` filters by id/goal substring). +- `POST /api/experiments` — register an experiment before its first arm. Body + `{ "id": "sb2", "goal": "..." }`; the id is the dash-free token job names + group under (`sb2-n8` → experiment `sb2`). +- `GET /api/experiments/:id` — arms, per-task matrix, and calibrated projections. +- `PUT /api/experiments/:id` — update the goal and per-run role/note/label. +- `POST /api/experiments/:id/arms` — launch a comparable arm; sample + config + inherited from a sibling. +- `DELETE /api/experiments/:id` — delete every arm (DB rows **and** job dirs) + plus the goal row; rejected while any arm is running. +- `GET /api/runs[?experiment=&status=&benchmark=]` — uniform run rows with + benchmark, score, progress, spend, and tokens. - `POST /api/runs` — launch through a benchmark adapter. Body: ```json @@ -51,7 +62,10 @@ bun run serve --port 4700 `include`, `timeoutMultiplier`, and `prewalk`; edit uses `include` as task IDs; SnapCompact uses `conditions` and treats `tasks` as the passage limit. - `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read). -- `DELETE /api/runs/:name` — cancel a manager-launched run. +- `POST /api/runs/:name/cancel` — cancel a manager-launched run. +- `DELETE /api/runs/:name` — permanently delete a finished run (DB row **and** + job dir; a surviving dir would be re-discovered on restart); rejected while + the run is live. - `POST /api/runs/:name/resume` — resume an incomplete harbor run in place: completed trials (and their spend) are reused, interrupted/pending trials re-run, and errored trials retried (body `{ "filterErrorTypes": [...] }` @@ -62,7 +76,7 @@ bun run serve --port 4700 - `GET /api/runs/:name/traces/:trace[?raw=1]` — normalized or native trace. - `GET /api/events` — SSE stream of run-list snapshots (sent on change). -State lives in `/_manager/harbor-manager.sqlite`; the filesystem +State lives in `/_manager/metaharness.sqlite`; the filesystem stays the source of truth and historical CLI runs are auto-discovered. ## Harbor runner options (excerpt) diff --git a/packages/harbor-manager/adapters/edit/bun-imports.d.ts b/packages/metaharness/adapters/edit/bun-imports.d.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/bun-imports.d.ts rename to packages/metaharness/adapters/edit/bun-imports.d.ts diff --git a/packages/harbor-manager/adapters/edit/cli.ts b/packages/metaharness/adapters/edit/cli.ts similarity index 98% rename from packages/harbor-manager/adapters/edit/cli.ts rename to packages/metaharness/adapters/edit/cli.ts index dbbe8ce04..eff9e91de 100644 --- a/packages/harbor-manager/adapters/edit/cli.ts +++ b/packages/metaharness/adapters/edit/cli.ts @@ -11,7 +11,7 @@ import { type BenchmarkConfig, runBenchmark } from "./runner"; const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark"); async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> { - const temp = await TempDir.create("@harbor-edit-fixtures-"); + const temp = await TempDir.create("@metaharness-edit-fixtures-"); const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer()); for (const [filePath, file] of await archive.files()) { await Bun.write(path.join(temp.path(), filePath), file); diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md b/packages/metaharness/adapters/edit/prompts/benchmark-retry.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md rename to packages/metaharness/adapters/edit/prompts/benchmark-retry.md diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-system.md b/packages/metaharness/adapters/edit/prompts/benchmark-system.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-system.md rename to packages/metaharness/adapters/edit/prompts/benchmark-system.md diff --git a/packages/harbor-manager/adapters/edit/prompts/benchmark-task.md b/packages/metaharness/adapters/edit/prompts/benchmark-task.md similarity index 100% rename from packages/harbor-manager/adapters/edit/prompts/benchmark-task.md rename to packages/metaharness/adapters/edit/prompts/benchmark-task.md diff --git a/packages/harbor-manager/adapters/edit/report.ts b/packages/metaharness/adapters/edit/report.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/report.ts rename to packages/metaharness/adapters/edit/report.ts diff --git a/packages/harbor-manager/adapters/edit/runner.test.ts b/packages/metaharness/adapters/edit/runner.test.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/runner.test.ts rename to packages/metaharness/adapters/edit/runner.test.ts diff --git a/packages/harbor-manager/adapters/edit/runner.ts b/packages/metaharness/adapters/edit/runner.ts similarity index 100% rename from packages/harbor-manager/adapters/edit/runner.ts rename to packages/metaharness/adapters/edit/runner.ts diff --git a/packages/harbor-manager/adapters/edit/tsconfig.json b/packages/metaharness/adapters/edit/tsconfig.json similarity index 100% rename from packages/harbor-manager/adapters/edit/tsconfig.json rename to packages/metaharness/adapters/edit/tsconfig.json diff --git a/packages/harbor-manager/agent/omp_local.py b/packages/metaharness/agent/omp_local.py similarity index 99% rename from packages/harbor-manager/agent/omp_local.py rename to packages/metaharness/agent/omp_local.py index c133fbfc5..46c4fb9d0 100644 --- a/packages/harbor-manager/agent/omp_local.py +++ b/packages/metaharness/agent/omp_local.py @@ -433,7 +433,7 @@ class OmpLocal(BaseInstalledAgent): ) def _generate_models_yaml(self) -> str: - lines = ["# Generated by harbor-manager runner — routes auth via host gateway.", "providers:"] + lines = ["# Generated by metaharness runner — routes auth via host gateway.", "providers:"] for provider in self._gateway_providers: lines += [ f" {provider}:", @@ -450,7 +450,7 @@ class OmpLocal(BaseInstalledAgent): web_search can't authenticate through the gateway, so it's off by default. """ lines = [ - "# Generated by harbor-manager runner.", + "# Generated by metaharness runner.", "web_search:", f" enabled: {'true' if self._web_search else 'false'}", ] diff --git a/packages/harbor-manager/package.json b/packages/metaharness/package.json similarity index 92% rename from packages/harbor-manager/package.json rename to packages/metaharness/package.json index 3b7e14fee..b50e9c348 100644 --- a/packages/harbor-manager/package.json +++ b/packages/metaharness/package.json @@ -1,7 +1,7 @@ { "type": "module", "private": true, - "name": "@oh-my-pi/harbor-manager", + "name": "@oh-my-pi/pi-metaharness", "version": "0.0.1", "description": "Unified benchmark runners plus Harbor run storage, REST/SSE APIs, and a live web dashboard", "homepage": "https://omp.sh", @@ -10,10 +10,10 @@ "repository": { "type": "git", "url": "git+https://github.com/can1357/oh-my-pi.git", - "directory": "packages/harbor-manager" + "directory": "packages/metaharness" }, "bin": { - "harbor-manager": "src/server.ts" + "metaharness": "src/server.ts" }, "scripts": { "check": "biome check . && bun run check:types", diff --git a/packages/harbor-manager/src/adapters/snapcompact.py b/packages/metaharness/src/adapters/snapcompact.py similarity index 100% rename from packages/harbor-manager/src/adapters/snapcompact.py rename to packages/metaharness/src/adapters/snapcompact.py diff --git a/packages/harbor-manager/src/benchmarks.test.ts b/packages/metaharness/src/benchmarks.test.ts similarity index 100% rename from packages/harbor-manager/src/benchmarks.test.ts rename to packages/metaharness/src/benchmarks.test.ts diff --git a/packages/harbor-manager/src/benchmarks.ts b/packages/metaharness/src/benchmarks.ts similarity index 100% rename from packages/harbor-manager/src/benchmarks.ts rename to packages/metaharness/src/benchmarks.ts diff --git a/packages/harbor-manager/src/experiments.test.ts b/packages/metaharness/src/experiments.test.ts similarity index 100% rename from packages/harbor-manager/src/experiments.test.ts rename to packages/metaharness/src/experiments.test.ts diff --git a/packages/harbor-manager/src/experiments.ts b/packages/metaharness/src/experiments.ts similarity index 94% rename from packages/harbor-manager/src/experiments.ts rename to packages/metaharness/src/experiments.ts index 423084c1f..5d5a59b1a 100644 --- a/packages/harbor-manager/src/experiments.ts +++ b/packages/metaharness/src/experiments.ts @@ -204,7 +204,7 @@ export function buildExperiments(store: RunStore): ExperimentSummary[] { for (const [id, runs] of groups) { out.push({ id, - goal: store.getExperimentGoal(id), + goal: store.getExperimentMeta(id)?.goal ?? "", arms: runs.length, runningArms: runs.filter(r => r.status === "running").length, datasets: [...new Set(runs.map(r => r.dataset).filter(Boolean))], @@ -218,6 +218,26 @@ export function buildExperiments(store: RunStore): ExperimentSummary[] { updatedAt: Math.max(...runs.map(r => r.finishedAt ?? Date.now())), }); } + // Registered-but-empty experiments (created via POST /api/experiments, no + // arms yet) are still browsable: zeroed rollups, goal from the meta row. + for (const meta of store.listExperimentMeta()) { + if (groups.has(meta.id)) continue; + out.push({ + id: meta.id, + goal: meta.goal, + arms: 0, + runningArms: 0, + datasets: [], + nTotal: 0, + done: 0, + pass: 0, + fail: 0, + error: 0, + costUsd: 0, + createdAt: meta.updatedAt, + updatedAt: meta.updatedAt, + }); + } out.sort((a, b) => b.updatedAt - a.updatedAt); return out; } @@ -259,7 +279,11 @@ export function pickMergedTrials(traces: TraceRow[]): TraceRow[] { export function experimentDetail(store: RunStore, id: string): ExperimentDetail | null { const runs = store.listRuns().filter(r => experimentOf(r.jobName) === id); - if (runs.length === 0) return null; + if (runs.length === 0) { + // Registered but armless (POST /api/experiments): still readable. + const meta = store.getExperimentMeta(id); + return meta ? { id, goal: meta.goal, arms: [], tasks: [], matrix: {} } : null; + } // One row per CANONICAL arm: `-fix`/`-backfill` re-runs merge into their // base arm — per-task best trial, summed spend. const groups = new Map(); @@ -347,5 +371,5 @@ export function experimentDetail(store: RunStore, id: string): ExperimentDetail // "reference rows, then treatments". const roleRank = (role: string) => (role === "baseline" ? 0 : role === "variant" ? 1 : 2); arms.sort((a, b) => roleRank(a.run.role) - roleRank(b.run.role) || a.arm.localeCompare(b.arm)); - return { id, goal: store.getExperimentGoal(id), arms, tasks: [...tasks].sort(), matrix }; + return { id, goal: store.getExperimentMeta(id)?.goal ?? "", arms, tasks: [...tasks].sort(), matrix }; } diff --git a/packages/harbor-manager/src/launch-args.ts b/packages/metaharness/src/launch-args.ts similarity index 100% rename from packages/harbor-manager/src/launch-args.ts rename to packages/metaharness/src/launch-args.ts diff --git a/packages/harbor-manager/src/manager.test.ts b/packages/metaharness/src/manager.test.ts similarity index 80% rename from packages/harbor-manager/src/manager.test.ts rename to packages/metaharness/src/manager.test.ts index 4f709d82d..70044461e 100644 --- a/packages/harbor-manager/src/manager.test.ts +++ b/packages/metaharness/src/manager.test.ts @@ -19,7 +19,7 @@ afterEach(() => { }); function makeJobsDir(): string { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-manager-test-")); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "metaharness-test-")); cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); return dir; } @@ -238,10 +238,13 @@ describe("ManagerServer API", () => { }); expect(badLaunch.status).toBe(400); - const cancelUnknown = (await (await fetch(`${base}/api/runs/nope`, { method: "DELETE" })).json()) as { + const cancelUnknown = (await (await fetch(`${base}/api/runs/nope/cancel`, { method: "POST" })).json()) as { cancelled: boolean; }; expect(cancelUnknown.cancelled).toBe(false); + + const deleteUnknown = await fetch(`${base}/api/runs/nope`, { method: "DELETE" }); + expect(deleteUnknown.status).toBe(404); }); it("serves edit and SnapCompact metrics and native traces through one API", async () => { @@ -366,6 +369,100 @@ describe("ManagerServer API", () => { expect(await resumeError("job-live")).toMatch(/already running/); expect(await resumeError("job-bare")).toMatch(/no harbor config.json/); }); + + it("experiment CRUD: create is browsable, delete removes rows + job dirs, live arms are protected", async () => { + const jobsDir = makeJobsDir(); + const manager = new ManagerServer(jobsDir); + // Two finished arms of experiment `crud` and one live run in a different experiment. + for (const jobName of ["crud-base", "crud-treat"]) { + manager.store.registerLaunch({ + benchmark: "harbor", + jobName, + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + manager.store.markExit(jobName, 0); + } + manager.store.registerLaunch({ + benchmark: "harbor", + jobName: "live-run", + dataset: "terminal-bench@2.0", + agent: "omp", + models: ["m/x"], + pid: process.pid, + }); + const server = manager.start(0); + cleanups.push(() => { + void manager.stop(); + }); + const base = `http://localhost:${server.port}`; + + // Create: registered id is browsable before any run exists. + const created = await fetch(`${base}/api/experiments`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ id: "fresh", goal: "does X beat Y?" }), + }); + expect(created.status).toBe(201); + const list = (await (await fetch(`${base}/api/experiments`)).json()) as Array<{ + id: string; + goal: string; + arms: number; + }>; + const fresh = list.find(e => e.id === "fresh"); + expect(fresh).toMatchObject({ goal: "does X beat Y?", arms: 0 }); + const freshDetail = (await (await fetch(`${base}/api/experiments/fresh`)).json()) as { + goal: string; + arms: unknown[]; + }; + expect(freshDetail).toMatchObject({ goal: "does X beat Y?", arms: [] }); + + // Create: dashed / empty ids can never own a run — rejected. + for (const id of ["bad-id", ""]) { + const res = await fetch(`${base}/api/experiments`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ id }), + }); + expect(res.status).toBe(400); + } + + // Browse: list filters. + const filtered = (await (await fetch(`${base}/api/runs?experiment=crud`)).json()) as Array<{ + jobName: string; + }>; + expect(filtered.map(r => r.jobName).sort()).toEqual(["crud-base", "crud-treat"]); + const running = (await (await fetch(`${base}/api/runs?status=running`)).json()) as Array<{ + jobName: string; + }>; + expect(running.map(r => r.jobName)).toEqual(["live-run"]); + const q = (await (await fetch(`${base}/api/experiments?q=fresh`)).json()) as Array<{ id: string }>; + expect(q.map(e => e.id)).toEqual(["fresh"]); + + // Delete run: live runs are protected, finished runs vanish from DB and disk. + const liveDelete = await fetch(`${base}/api/runs/live-run`, { method: "DELETE" }); + expect(liveDelete.status).toBe(400); + const runDelete = await fetch(`${base}/api/runs/crud-treat`, { method: "DELETE" }); + expect(runDelete.status).toBe(200); + expect(fs.existsSync(path.join(jobsDir, "crud-treat"))).toBe(false); + expect(manager.store.getRun("crud-treat")).toBeNull(); + + // Delete experiment: remaining arm rows + dirs + goal row all go; 404 after. + const expDelete = (await (await fetch(`${base}/api/experiments/crud`, { method: "DELETE" })).json()) as { + deletedRuns: string[]; + }; + expect(expDelete.deletedRuns).toEqual(["crud-base"]); + expect(fs.existsSync(path.join(jobsDir, "crud-base"))).toBe(false); + expect((await fetch(`${base}/api/experiments/crud`)).status).toBe(404); + expect((await fetch(`${base}/api/experiments/unknown`, { method: "DELETE" })).status).toBe(404); + + // Delete experiment with a live arm: refused, nothing removed. + const liveExpDelete = await fetch(`${base}/api/experiments/live`, { method: "DELETE" }); + expect(liveExpDelete.status).toBe(400); + expect(manager.store.getRun("live-run")).not.toBeNull(); + }); }); describe("resolveArmLaunch", () => { diff --git a/packages/harbor-manager/src/runner.test.ts b/packages/metaharness/src/runner.test.ts similarity index 100% rename from packages/harbor-manager/src/runner.test.ts rename to packages/metaharness/src/runner.test.ts diff --git a/packages/harbor-manager/src/runner.ts b/packages/metaharness/src/runner.ts similarity index 99% rename from packages/harbor-manager/src/runner.ts rename to packages/metaharness/src/runner.ts index b832a3d63..1bb8ac385 100755 --- a/packages/harbor-manager/src/runner.ts +++ b/packages/metaharness/src/runner.ts @@ -14,9 +14,9 @@ import * as path from "node:path"; * process renders a live dashboard (progress / success% / spend / tokens / ETA) * by polling each trial's `result.json`. On completion it writes a markdown report. * - * harbor-manager harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 - * harbor-manager harbor --agent oracle --tasks 2 # cheap pipeline smoke - * harbor-manager harbor --help + * metaharness harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 + * metaharness harbor --agent oracle --tasks 2 # cheap pipeline smoke + * metaharness harbor --help */ import type { Server } from "bun"; import { harborRunnerArgs, type LaunchRequest } from "./launch-args"; @@ -128,9 +128,9 @@ function defaultConfig(): Config { }; } -const HELP = `harbor-manager runner (local omp) +const HELP = `metaharness runner (local omp) -Usage: harbor-manager harbor [options] [-- ] +Usage: metaharness harbor [options] [-- ] Commands: cleanup Force-remove ALL leftover Harbor containers + networks, then exit @@ -1240,7 +1240,7 @@ function deriveProviders(cfg: Config): string[] { function writeModelsYaml(benchDir: string, cfg: Config): string { const providers = deriveProviders(cfg); - const lines = ["# Generated by harbor-manager — auth via host pm2 gateway.", "providers:"]; + const lines = ["# Generated by metaharness — auth via host pm2 gateway.", "providers:"]; for (const p of providers) { lines.push(` ${p}:`); lines.push(` baseUrl: ${cfg.gatewayUrl}`); diff --git a/packages/harbor-manager/src/server.ts b/packages/metaharness/src/server.ts similarity index 79% rename from packages/harbor-manager/src/server.ts rename to packages/metaharness/src/server.ts index 9c8045d51..17d31b513 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/metaharness/src/server.ts @@ -1,16 +1,23 @@ #!/usr/bin/env bun /** - * harbor-manager server: REST + SSE API over the run store, static web + * metaharness server: REST + SSE API over the run store, static web * dashboard, and a launcher that spawns the CLI runner as a managed child. * * bun src/server.ts [--port 4700] [--jobs-dir ] * * API: - * GET /api/experiments → experiment summaries across all benchmarks - * GET /api/runs → RunRow[] + * GET /api/experiments[?q=] → experiment summaries across all benchmarks + * POST /api/experiments → register an experiment (id + goal) before its first arm + * GET /api/experiments/:id → experiment detail (arms, task matrix) + * PUT /api/experiments/:id → update goal + per-run role/note/label + * DELETE /api/experiments/:id → delete all arms (rows + job dirs) and the goal row + * POST /api/experiments/:id/arms → launch a comparable arm + * GET /api/runs[?experiment=&status=&benchmark=] → RunRow[] * POST /api/runs → launch any benchmark * GET /api/runs/:name → { run, traces } - * DELETE /api/runs/:name → cancel a managed run + * POST /api/runs/:name/cancel → cancel a managed run + * POST /api/runs/:name/resume → resume an incomplete harbor run + * DELETE /api/runs/:name → delete a finished run (row + job dir) * GET /api/runs/:name/traces/:trace → normalized trace * GET /api/events → SSE: run-list snapshots on change */ @@ -28,6 +35,13 @@ export interface ExperimentMetaUpdate { runs?: Record; } +/** POST /api/experiments body — pre-registers an experiment id with a goal. */ +export interface CreateExperimentRequest { + /** Dash-free token; runs group into it as `-` job names. */ + id: string; + goal?: string; +} + import indexHtml from "./web/index.html"; const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); @@ -76,6 +90,24 @@ function parseServerArgs(argv: string[]): { port: number; jobsDir: string } { return { port, jobsDir }; } +/** Job names are single path segments; anything else could escape the jobs dir. */ +function assertSafeJobName(jobName: string): void { + if (!jobName || jobName === "." || jobName === ".." || /[/\\]/.test(jobName)) { + throw new Error(`invalid job name: ${jobName}`); + } +} + +/** True when `pid` names a live process (signal-0 probe). */ +function pidAlive(pid: number | null): boolean { + if (pid == null) return false; + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +} + /** * Resolve the launch request for a new arm added to an existing experiment. * Inherits the experiment's benchmark, dataset, and — crucially — the exact @@ -236,7 +268,17 @@ export class ManagerServer { return Response.json(BENCHMARK_DEFINITIONS); } if (p === "/api/experiments" && request.method === "GET") { - return Response.json(buildExperiments(this.#store)); + const q = url.searchParams.get("q")?.toLowerCase() ?? ""; + const experiments = buildExperiments(this.#store); + return Response.json( + q + ? experiments.filter(e => e.id.toLowerCase().includes(q) || e.goal.toLowerCase().includes(q)) + : experiments, + ); + } + if (p === "/api/experiments" && request.method === "POST") { + const body = (await request.json()) as CreateExperimentRequest; + return Response.json(this.createExperiment(body), { status: 201 }); } const expMatch = p.match(/^\/api\/experiments\/([^/]+)$/); if (expMatch) { @@ -245,6 +287,11 @@ export class ManagerServer { const body = (await request.json()) as ExperimentMetaUpdate; return Response.json(this.updateExperimentMeta(id, body)); } + if (request.method === "DELETE") { + const result = this.deleteExperiment(id); + if (!result) return Response.json({ error: "experiment not found" }, { status: 404 }); + return Response.json(result); + } const detail = experimentDetail(this.#store, id); if (!detail) return Response.json({ error: "experiment not found" }, { status: 404 }); return Response.json(detail); @@ -256,7 +303,14 @@ export class ManagerServer { return Response.json(this.addArm(id, body), { status: 201 }); } if (p === "/api/runs" && request.method === "GET") { - return Response.json(this.#store.listRuns()); + const experiment = url.searchParams.get("experiment"); + const status = url.searchParams.get("status"); + const benchmark = url.searchParams.get("benchmark"); + let runs = this.#store.listRuns(); + if (experiment) runs = runs.filter(r => experimentOf(r.jobName) === experiment); + if (status) runs = runs.filter(r => r.status === status); + if (benchmark) runs = runs.filter(r => r.benchmark === benchmark); + return Response.json(runs); } if (p === "/api/runs" && request.method === "POST") { const body = (await request.json()) as LaunchRequest; @@ -268,10 +322,17 @@ export class ManagerServer { const body = (await request.json().catch(() => ({}))) as { filterErrorTypes?: string[] }; return Response.json(this.resume(jobName, body), { status: 201 }); } + const cancelMatch = p.match(/^\/api\/runs\/([^/]+)\/cancel$/); + if (cancelMatch && request.method === "POST") { + return Response.json(this.cancel(decodeURIComponent(cancelMatch[1]))); + } const runMatch = p.match(/^\/api\/runs\/([^/]+)$/); if (runMatch) { const jobName = decodeURIComponent(runMatch[1]); - if (request.method === "DELETE") return Response.json(this.cancel(jobName)); + if (request.method === "DELETE") { + if (!this.deleteRun(jobName)) return Response.json({ error: "run not found" }, { status: 404 }); + return Response.json({ jobName, deleted: true }); + } const run = this.#store.syncRun(jobName); if (!run) return Response.json({ error: "run not found" }, { status: 404 }); return Response.json({ run, traces: this.#store.listTraces(jobName) }); @@ -386,16 +447,7 @@ export class ManagerServer { // Trust liveness, not the recorded status: a runner killed while a // previous server instance owned it leaves a stale `running` row with a // dead (or null) pid and nobody to fire markExit. - const pidAlive = (pid: number | null): boolean => { - if (pid == null) return false; - try { - process.kill(pid, 0); - return true; - } catch { - return false; - } - }; - if (this.#children.has(jobName) || (run.status === "running" && pidAlive(run.pid))) { + if (this.#runLive(run)) { throw new Error(`run ${jobName} is already running`); } if (run.status === "running") this.#store.markExit(jobName, null, true); @@ -461,18 +513,80 @@ export class ManagerServer { return proc.pid; } + /** Liveness check that survives manager restarts: managed child, or a running row with a live pid. */ + #runLive(run: RunRow): boolean { + return this.#children.has(run.jobName) || (run.status === "running" && pidAlive(run.pid)); + } + + /** Register an experiment id (with an optional goal) so it is browsable before its first arm. */ + createExperiment(req: CreateExperimentRequest): { id: string; goal: string } { + const id = req.id?.trim() ?? ""; + // Dashes are structurally impossible: `experimentOf` groups job names by + // the token before the first dash, so a dashed id could never own a run. + if (!/^[A-Za-z0-9_.]+$/.test(id)) { + throw new Error("experiment id must be a non-empty token of [A-Za-z0-9_.] (runs group as `-`)"); + } + const goal = req.goal ?? this.#store.getExperimentMeta(id)?.goal ?? ""; + this.#store.setExperimentGoal(id, goal); + return { id, goal }; + } + /** Apply goal + per-run role/note metadata; used by the UI and for backfill. */ updateExperimentMeta(id: string, update: ExperimentMetaUpdate): { id: string; updatedRuns: string[] } { if (update.goal !== undefined) this.#store.setExperimentGoal(id, update.goal); const updatedRuns: string[] = []; - for (const [jobName, meta] of Object.entries(update.runs ?? {})) { + for (const jobName in update.runs) { if (experimentOf(jobName) !== id) continue; - if (this.#store.setRunMeta(jobName, meta)) updatedRuns.push(jobName); + if (this.#store.setRunMeta(jobName, update.runs[jobName])) updatedRuns.push(jobName); } this.#tick(); return { id, updatedRuns }; } + /** + * Delete an experiment: every arm's DB row, job dir, and manager log, plus + * the goal row. Refuses while any arm is live (cancel first — deleting a + * job dir under a writing runner would corrupt it). Returns null when the + * id names neither runs nor a registered experiment. + */ + deleteExperiment(id: string): { id: string; deletedRuns: string[] } | null { + const runs = this.#store.listRuns().filter(r => experimentOf(r.jobName) === id); + if (runs.length === 0 && !this.#store.getExperimentMeta(id)) return null; + const live = runs.filter(r => this.#runLive(r)); + if (live.length > 0) { + throw new Error( + `experiment ${id} has running arms (${live.map(r => r.jobName).join(", ")}); cancel them first`, + ); + } + for (const run of runs) this.#destroyRun(run.jobName); + this.#store.deleteExperimentMeta(id); + this.#tick(); + return { id, deletedRuns: runs.map(r => r.jobName) }; + } + + /** + * Permanently delete a run: DB row + trials, job dir, and manager log. + * Disk removal is not optional — discover() would resurrect a surviving + * job dir as a fresh row on the next restart. Refuses while the run is + * live; returns false when the run is unknown. + */ + deleteRun(jobName: string): boolean { + const run = this.#store.getRun(jobName); + if (!run) return false; + if (this.#runLive(run)) throw new Error(`run ${jobName} is running; cancel it first`); + this.#destroyRun(jobName); + this.#tick(); + return true; + } + + /** Remove a run's DB rows and on-disk artifacts (job dir + manager log). */ + #destroyRun(jobName: string): void { + assertSafeJobName(jobName); + this.#store.deleteRun(jobName); + fs.rmSync(path.join(this.jobsDir, jobName), { recursive: true, force: true }); + fs.rmSync(path.join(this.jobsDir, "_manager", "logs", `${jobName}.log`), { force: true }); + } + /** Add a comparable arm to an existing experiment, inheriting its sample + config. */ addArm(experimentId: string, req: AddArmRequest): { jobName: string; pid: number } { return this.launch(resolveArmLaunch(this.#store, experimentId, req)); @@ -642,20 +756,20 @@ if (import.meta.main) { // `bun --hot` re-evaluates this module in-place: retire the previous // instance first, or its sync ticker and sqlite connection leak per reload. const host = globalThis as typeof globalThis & { - __harborManagerServer?: ManagerServer; - __harborManagerHooks?: boolean; + __metaharnessServer?: ManagerServer; + __metaharnessHooks?: boolean; }; - await host.__harborManagerServer?.stop(); + await host.__metaharnessServer?.stop(); const { port, jobsDir } = parseServerArgs(process.argv.slice(2)); const manager = new ManagerServer(jobsDir); - host.__harborManagerServer = manager; + host.__metaharnessServer = manager; const server = manager.start(port); - process.stdout.write(`harbor-manager listening on http://localhost:${server.port} (jobs: ${jobsDir})\n`); + process.stdout.write(`metaharness listening on http://localhost:${server.port} (jobs: ${jobsDir})\n`); // Process-wide hooks register once; `--hot` re-evals reuse them via `host`. - if (!host.__harborManagerHooks) { - host.__harborManagerHooks = true; + if (!host.__metaharnessHooks) { + host.__metaharnessHooks = true; const shutdown = async () => { - await host.__harborManagerServer?.stop(); + await host.__metaharnessServer?.stop(); process.exit(0); }; process.on("SIGINT", shutdown); diff --git a/packages/harbor-manager/src/store.ts b/packages/metaharness/src/store.ts similarity index 92% rename from packages/harbor-manager/src/store.ts rename to packages/metaharness/src/store.ts index c57d96670..c06f24fc7 100644 --- a/packages/harbor-manager/src/store.ts +++ b/packages/metaharness/src/store.ts @@ -72,6 +72,13 @@ export interface TraceRow { tracePath: string | null; } +/** Row in the `experiments` table: goal metadata keyed by experiment id. */ +export interface ExperimentMeta { + id: string; + goal: string; + updatedAt: number; +} + export interface LaunchRecord { benchmark: BenchmarkKind; jobName: string; @@ -180,7 +187,7 @@ export class RunStore { constructor(jobsDir: string, dbPath?: string) { this.jobsDir = jobsDir; fs.mkdirSync(path.join(jobsDir, "_manager"), { recursive: true }); - this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite")); + this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "metaharness.sqlite")); this.#db.run("PRAGMA busy_timeout = 5000"); enableWal(this.#db); this.#db.run(SCHEMA); @@ -258,9 +265,39 @@ export class RunStore { .run(id, goal, Date.now()); } - getExperimentGoal(id: string): string { - const row = this.#db.query("SELECT goal FROM experiments WHERE id = ?").get(id) as { goal: string } | null; - return row?.goal ?? ""; + /** Stored experiment metadata, or null when the id was never registered. */ + getExperimentMeta(id: string): ExperimentMeta | null { + const row = this.#db.query("SELECT id, goal, updated_at FROM experiments WHERE id = ?").get(id) as { + id: string; + goal: string; + updated_at: number; + } | null; + return row ? { id: row.id, goal: row.goal, updatedAt: row.updated_at } : null; + } + + /** Every registered experiment row, newest first. */ + listExperimentMeta(): ExperimentMeta[] { + const rows = this.#db + .query("SELECT id, goal, updated_at FROM experiments ORDER BY updated_at DESC") + .all() as Array<{ + id: string; + goal: string; + updated_at: number; + }>; + return rows.map(r => ({ id: r.id, goal: r.goal, updatedAt: r.updated_at })); + } + + /** Drop the experiment metadata row (run rows are deleted separately via deleteRun). */ + deleteExperimentMeta(id: string): void { + this.#db.query("DELETE FROM experiments WHERE id = ?").run(id); + } + + /** Delete a run row and its trials; returns false when the run is unknown. */ + deleteRun(jobName: string): boolean { + if (!this.getRun(jobName)) return false; + this.#db.query("DELETE FROM trials WHERE job_name = ?").run(jobName); + this.#db.query("DELETE FROM runs WHERE job_name = ?").run(jobName); + return true; } /** Set role/note/label metadata on an existing run row. */ diff --git a/packages/harbor-manager/src/web/app.tsx b/packages/metaharness/src/web/app.tsx similarity index 99% rename from packages/harbor-manager/src/web/app.tsx rename to packages/metaharness/src/web/app.tsx index 2b3318a60..3afe3a0be 100644 --- a/packages/harbor-manager/src/web/app.tsx +++ b/packages/metaharness/src/web/app.tsx @@ -1,5 +1,5 @@ /** - * harbor-manager dashboard. + * metaharness dashboard. * * Views (hash-routed): * #/ experiments index — runs grouped by job-name prefix @@ -1696,7 +1696,7 @@ function RunsPage({ selected }: { selected: string | null }) { if (el) el.scrollTop = el.scrollHeight; }, [traceData]); const cancel = useCallback(async (name: string) => { - if (confirm(`stop ${name}?`)) await fetch(`/api/runs/${encodeURIComponent(name)}`, { method: "DELETE" }); + if (confirm(`stop ${name}?`)) await fetch(`/api/runs/${encodeURIComponent(name)}/cancel`, { method: "POST" }); }, []); const resume = useCallback(async (name: string) => { if (!confirm(`resume ${name}? completed trials are kept; interrupted, pending, and errored ones re-run`)) return; @@ -1966,7 +1966,7 @@ function App() { return ( <>
-

harbor-manager

+

metaharness