From 31ae1b96fd27a39fb2e085d0a1cbc84b66403661 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 2 May 2026 03:31:27 +0200 Subject: [PATCH] feat(coding-agent/autoresearch): added branch-specific state restore - Added branch-aware session loading so autoresearch state only rehydrates for current branch. - Replaced user-specified experiment commands with fixed `bash autoresearch.sh` execution flow. - Enforced safer setup checks, including missing `autoresearch.sh` and uncommitted-worktree errors. - Added branch-specific storage helpers, baseline-commit persistence, and expanded tests for dirty-path cases. --- packages/coding-agent/src/autoresearch/git.ts | 6 +- .../coding-agent/src/autoresearch/index.ts | 119 ++++++-- .../src/autoresearch/prompt-setup.md | 43 +++ .../coding-agent/src/autoresearch/prompt.md | 11 +- .../coding-agent/src/autoresearch/state.ts | 2 - .../coding-agent/src/autoresearch/storage.ts | 67 ++++- .../src/autoresearch/tools/init-experiment.ts | 107 +++++-- .../src/autoresearch/tools/log-experiment.ts | 42 ++- .../src/autoresearch/tools/run-experiment.ts | 38 ++- .../src/autoresearch/tools/update-notes.ts | 12 +- .../coding-agent/src/autoresearch/types.ts | 2 - .../test/autoresearch-tools.test.ts | 263 +++++++++++++++--- 12 files changed, 556 insertions(+), 156 deletions(-) create mode 100644 packages/coding-agent/src/autoresearch/prompt-setup.md diff --git a/packages/coding-agent/src/autoresearch/git.ts b/packages/coding-agent/src/autoresearch/git.ts index 81995dea7..95c54700a 100644 --- a/packages/coding-agent/src/autoresearch/git.ts +++ b/packages/coding-agent/src/autoresearch/git.ts @@ -67,10 +67,8 @@ export async function ensureAutoresearchBranch( if (dirtyPaths.length > 0) { const preview = formatDirtyPaths(dirtyPaths); return { - ok: true, - branchName: null, - created: false, - warning: `Worktree is dirty (${preview}). Continuing on the current branch; auto-commit and full-tree reset are disabled until you commit/stash these changes.`, + ok: false, + error: `Worktree is dirty (${preview}). Commit or stash these changes before starting autoresearch — a fresh autoresearch/* branch needs a clean baseline.`, }; } diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 407cac3c2..9491ffb78 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -9,6 +9,7 @@ import { createDashboardController } from "./dashboard"; import { ensureAutoresearchBranch } from "./git"; import { formatNum } from "./helpers"; import promptTemplate from "./prompt.md" with { type: "text" }; +import setupPromptTemplate from "./prompt-setup.md" with { type: "text" }; import resumeMessageTemplate from "./resume-message.md" with { type: "text" }; import { buildExperimentState, @@ -20,7 +21,7 @@ import { findBestKeptMetric, reconstructControlState, } from "./state"; -import { openAutoresearchStorage, type RunRow } from "./storage"; +import { openAutoresearchStorage, openAutoresearchStorageIfExists, type RunRow, type SessionRow } from "./storage"; import { createInitExperimentTool } from "./tools/init-experiment"; import { createLogExperimentTool } from "./tools/log-experiment"; import { createRunExperimentTool } from "./tools/run-experiment"; @@ -36,21 +37,49 @@ export const createAutoresearchExtension: ExtensionFactory = api => { const getSessionKey = (ctx: ExtensionContext): string => ctx.sessionManager.getSessionId(); const getRuntime = (ctx: ExtensionContext): AutoresearchRuntime => runtimeStore.ensure(getSessionKey(ctx)); + const loadActiveSession = async ( + ctx: ExtensionContext, + ): Promise<{ session: SessionRow | null; currentBranch: string | null }> => { + const currentBranch = await tryReadBranch(ctx.cwd); + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + if (!storage) return { session: null, currentBranch }; + const session = storage.getActiveSessionForBranch(currentBranch); + return { session, currentBranch }; + }; + const rehydrate = async (ctx: ExtensionContext): Promise => { const runtime = getRuntime(ctx); const control = reconstructControlState(ctx.sessionManager.getBranch()); runtime.goal = control.goal; - runtime.autoresearchMode = control.autoresearchMode; runtime.autoResumeArmed = false; runtime.lastAutoResumePendingRunNumber = null; - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - if (session) { - const loggedRuns = storage.listLoggedRuns(session.id); - runtime.state = buildExperimentState(session, loggedRuns); - runtime.goal = runtime.goal ?? session.goal; - runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id)); + // Skip storage entirely if autoresearch was never activated in this conversation. + // This is the common case: every project gets a session_start event but most + // never touch autoresearch, so we must not create a SQLite file just to look. + const everActivated = control.lastMode !== null; + const { session, currentBranch } = everActivated + ? await loadActiveSession(ctx) + : { session: null, currentBranch: null }; + + // Mode is effective only when the recorded session matches the current git + // branch. When the user switches off the autoresearch branch the widget hides + // and the experiment tools detach, but the session entries are preserved so + // switching back resumes seamlessly. + const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch; + runtime.autoresearchMode = control.autoresearchMode && onActiveBranch; + + if (session && onActiveBranch) { + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + if (storage) { + const loggedRuns = storage.listLoggedRuns(session.id); + runtime.state = buildExperimentState(session, loggedRuns); + runtime.goal = runtime.goal ?? session.goal; + runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id)); + } else { + runtime.state = createExperimentState(); + runtime.lastRunSummary = null; + } } else { runtime.state = createExperimentState(); runtime.lastRunSummary = null; @@ -151,8 +180,12 @@ export const createAutoresearchExtension: ExtensionFactory = api => { ctx.ui.notify(branchResult.warning, "warning"); } - const storage = await openAutoresearchStorage(ctx.cwd); - const existingSession = storage.getActiveSession(); + // Look up an existing session for the branch we just landed on. A session + // recorded under a different autoresearch/* branch is intentionally ignored + // — `/autoresearch` on a fresh branch starts a fresh session. Only open the + // DB if it already exists; the empty-state path must not create one. + const existingStorage = await openAutoresearchStorageIfExists(ctx.cwd); + const existingSession = existingStorage?.getActiveSessionForBranch(branchResult.branchName) ?? null; const resumeContext = trimmed; const branchStatusLine = branchResult.branchName ? branchResult.created @@ -160,13 +193,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => { : `Using dedicated git branch \`${branchResult.branchName}\`.` : "Continuing on the current branch — no autoresearch branch was created."; - if (existingSession) { - if (goalArg) storage.updateSession(existingSession.id, { goal: goalArg }); + if (existingSession && existingStorage) { + if (goalArg) existingStorage.updateSession(existingSession.id, { goal: goalArg }); if (branchResult.branchName) { - storage.updateSession(existingSession.id, { branch: branchResult.branchName }); + existingStorage.updateSession(existingSession.id, { branch: branchResult.branchName }); } - const refreshed = storage.getSessionById(existingSession.id) ?? existingSession; - runtime.state = buildExperimentState(refreshed, storage.listLoggedRuns(refreshed.id)); + const refreshed = existingStorage.getSessionById(existingSession.id) ?? existingSession; + runtime.state = buildExperimentState(refreshed, existingStorage.listLoggedRuns(refreshed.id)); runtime.goal = refreshed.goal ?? goalArg; setMode(ctx, true, runtime.goal, "on"); dashboard.updateWidget(ctx, runtime); @@ -231,9 +264,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.autoResumeArmed = false; return; } - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - const pendingRow = session ? storage.getPendingRun(session.id) : null; + const { session } = await loadActiveSession(ctx); + const storage = session ? await openAutoresearchStorageIfExists(ctx.cwd) : null; + const pendingRow = session && storage ? storage.getPendingRun(session.id) : null; const pendingRun = pendingRunSummaryFromRow(pendingRow); runtime.lastRunSummary = pendingRun; runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; @@ -261,12 +294,27 @@ export const createAutoresearchExtension: ExtensionFactory = api => { api.on("before_agent_start", async (event, ctx) => { const runtime = getRuntime(ctx); if (!runtime.autoresearchMode) return; - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - if (session) { + // Re-check git branch on every agent start. If the user manually switched + // off the autoresearch/* branch between turns, we silently drop autoresearch + // from this turn — the widget hides, the experiment tools detach, and we do + // not inject the autoresearch system prompt. + const { session, currentBranch } = await loadActiveSession(ctx); + const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch; + if (!onActiveBranch) { + runtime.autoresearchMode = false; + runtime.state = createExperimentState(); + runtime.lastRunSummary = null; + runtime.runningExperiment = null; + dashboard.updateWidget(ctx, runtime); + const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); + await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); + return; + } + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + if (session && storage) { runtime.state = buildExperimentState(session, storage.listLoggedRuns(session.id)); } - const pendingRow = session ? storage.getPendingRun(session.id) : null; + const pendingRow = session && storage ? storage.getPendingRun(session.id) : null; const pendingRun = pendingRunSummaryFromRow(pendingRow); runtime.lastRunSummary = pendingRun; runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; @@ -301,9 +349,25 @@ export const createAutoresearchExtension: ExtensionFactory = api => { run_number: r.runNumber, paths: r.scopeDeviations.join(", "), })); - const lastCommand = pendingRun?.command ?? null; - const showCommandWarning = - Boolean(state.benchmarkCommand) && lastCommand !== null && lastCommand !== state.benchmarkCommand; + if (!session) { + const currentBranch = await tryReadBranch(ctx.cwd); + const onAutoresearchBranch = currentBranch?.startsWith("autoresearch/") ?? false; + const baselineWarning = onAutoresearchBranch + ? null + : "Heads up: you are not on a dedicated `autoresearch/*` branch. `log_experiment discard` will only revert run-modified files, not reset to baseline — so harness files written before `init_experiment` may not survive a discard. Clean the worktree and re-run `/autoresearch` if you want full revert safety."; + return { + systemPrompt: prompt.render(setupPromptTemplate, { + base_system_prompt: event.systemPrompt, + has_goal: goal.trim().length > 0, + goal, + working_dir: ctx.cwd, + has_branch: Boolean(currentBranch), + branch: currentBranch ?? "", + has_baseline_warning: baselineWarning !== null, + baseline_warning: baselineWarning ?? "", + }), + }; + } return { systemPrompt: prompt.render(promptTemplate, { base_system_prompt: event.systemPrompt, @@ -339,9 +403,6 @@ export const createAutoresearchExtension: ExtensionFactory = api => { pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined ? formatNum(pendingRun.parsedPrimary, state.metricUnit) : null, - has_preferred_command_warning: showCommandWarning, - preferred_command: state.benchmarkCommand ?? "", - last_command: lastCommand ?? "", }), }; }); diff --git a/packages/coding-agent/src/autoresearch/prompt-setup.md b/packages/coding-agent/src/autoresearch/prompt-setup.md new file mode 100644 index 000000000..e176ff45d --- /dev/null +++ b/packages/coding-agent/src/autoresearch/prompt-setup.md @@ -0,0 +1,43 @@ +{{base_system_prompt}} + +## Autoresearch Mode — Phase 1: Harness Setup + +Autoresearch mode is active and there is no session yet. Your job in this turn is to **build the benchmark harness**, not to optimise anything. Optimisation starts only after you call `init_experiment`. + +{{#if has_goal}} +Primary goal (for context — implement the harness so it can measure this): +{{goal}} +{{else}} +There is no goal recorded yet. Infer what to optimise from the latest user message and design the harness to measure that. Capture the goal when you call `init_experiment`. +{{/if}} + +Working directory: `{{working_dir}}` +{{#if has_branch}}Active branch: `{{branch}}`{{/if}} +{{#if has_baseline_warning}} + +{{baseline_warning}} +{{/if}} + +### What you must produce + +Write `./autoresearch.sh` at the working directory. It is the canonical benchmark entrypoint and must: + +- exit 0 on success and non-zero on failure; +- print the primary metric as a single line `METRIC =`; +- print any secondary metrics as additional `METRIC =` lines; +- run the same workload deterministically every time (no live network, no time-of-day dependencies, fixed seeds where applicable). + +You **may** edit anything else needed to make `autoresearch.sh` work — benchmark binaries, `Cargo.toml`, `package.json`, helper scripts, fixtures. All those edits are part of the harness baseline and will be committed for you when you call `init_experiment` on an autoresearch branch. + +### Steps + +1. Inspect the target. Read source, identify what to measure, decide on the workload. +2. Write `autoresearch.sh` plus any supporting files (benchmark binaries, fixtures, etc.). +3. Validate it: invoke `bash autoresearch.sh` through the regular `bash` tool. Confirm it exits 0 and emits at least one `METRIC` line. Iterate on the harness until it does. +4. Call `init_experiment` with the goal, primary metric (matching the `METRIC` name), and scope. This snapshots the worktree as the baseline and starts Phase 2 (the iteration loop). + +### Rules + +- Do **not** call `run_experiment`, `log_experiment`, or `update_notes` yet. They will error with "no active autoresearch session" until `init_experiment` runs. +- Do **not** treat a compile-only check as a benchmark. The harness must actually execute the workload and emit `METRIC`. +- Do **not** create `autoresearch.md`, `autoresearch.checks.sh`, `autoresearch.program.md`, `autoresearch.ideas.md`, `autoresearch.jsonl`, `.autoresearch/`, or `autoresearch.config.json`. Session state is tracked for you. diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md index 380e6d4db..da25c46a8 100644 --- a/packages/coding-agent/src/autoresearch/prompt.md +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -11,7 +11,7 @@ Primary goal: There is no goal recorded for this session yet. Infer what to optimize from the latest user message and the conversation; capture the goal in your notes (`update_notes`) once it is clear. {{/if}} -Session state and run artifacts are managed for you. Do not create `autoresearch.md`, `autoresearch.sh`, or `.autoresearch/` in this repo. +Session state and run artifacts are managed for you. The benchmark entrypoint is `bash autoresearch.sh` (committed during Phase 1). Do not edit `autoresearch.sh` mid-segment unless you intentionally bump segment via `init_experiment new_segment: true`. Do not create `autoresearch.md` or `.autoresearch/` in this repo. Working directory: `{{working_dir}}` {{#if has_branch}}Active branch: `{{branch}}`{{/if}} @@ -21,13 +21,13 @@ You are running an autonomous experiment loop. Keep iterating until the user int ### Available tools - `init_experiment` — open or reconfigure the session. Pass `new_segment: true` to start a fresh baseline within the current session. -- `run_experiment` — run any benchmark command. Pass the actual command; output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the command are parsed back to you. +- `run_experiment` — run the benchmark (`bash autoresearch.sh`). Output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the harness are parsed back to you. The command is fixed; if you need a different workload, edit `autoresearch.sh` and bump segment via `init_experiment new_segment: true`. - `log_experiment` — record the result. On `keep`, modified files are committed for you; on `discard`/`crash`/`checks_failed`, the worktree is reverted. Pass `flag_runs` to mark earlier runs as suspect; flagged runs are excluded from baseline and best-metric math. - `update_notes` — replace the durable session playbook (`body`) or append to the ideas backlog (`append_idea`). The notes are injected into your system prompt every iteration. ### Operating protocol 1. Understand the target before touching code: read source, identify the bottleneck, verify prerequisites and benchmark inputs. -2. Capture goal, benchmark command, primary metric, scope, and constraints in `init_experiment`. Update them later via another `init_experiment` call (no segment bump) or via `update_notes`. +2. Update goal, scope, or constraints via another `init_experiment` call (no segment bump) or `update_notes`. Bump segment when you intentionally change `autoresearch.sh`. 3. Establish a baseline first. 4. Iterate: change code, run `run_experiment`, log honestly with `log_experiment`. One coherent experiment per iteration. 5. Keep the primary metric as the decision maker: @@ -95,11 +95,6 @@ An unlogged run is waiting: Finish the `log_experiment` step before starting another benchmark. {{/if}} -{{#if has_preferred_command_warning}} - -### Preferred command -Last `run_experiment` used `{{last_command}}`. Preferred command for this segment is `{{preferred_command}}`. If the workload changed intentionally, call `init_experiment` to update the preferred command (or pass `new_segment: true` to start a fresh baseline). -{{/if}} ### Guardrails - Do not game the benchmark. diff --git a/packages/coding-agent/src/autoresearch/state.ts b/packages/coding-agent/src/autoresearch/state.ts index f159ac611..667f9e8be 100644 --- a/packages/coding-agent/src/autoresearch/state.ts +++ b/packages/coding-agent/src/autoresearch/state.ts @@ -26,7 +26,6 @@ export function createExperimentState(): ExperimentState { currentSegment: 0, maxExperiments: null, confidence: null, - benchmarkCommand: null, scopePaths: [], offLimits: [], constraints: [], @@ -177,7 +176,6 @@ export function buildExperimentState(session: SessionRow, loggedRuns: RunRow[]): state.metricName = session.primaryMetric; state.metricUnit = session.metricUnit; state.bestDirection = session.direction; - state.benchmarkCommand = session.preferredCommand; state.scopePaths = [...session.scopePaths]; state.offLimits = [...session.offLimits]; state.constraints = [...session.constraints]; diff --git a/packages/coding-agent/src/autoresearch/storage.ts b/packages/coding-agent/src/autoresearch/storage.ts index e85e4d3e0..3b5b72f10 100644 --- a/packages/coding-agent/src/autoresearch/storage.ts +++ b/packages/coding-agent/src/autoresearch/storage.ts @@ -1,7 +1,7 @@ import { Database, type SQLQueryBindings } from "bun:sqlite"; import * as fs from "node:fs"; import * as path from "node:path"; -import { getAutoresearchDbPath, getAutoresearchDir, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils"; +import { getAutoresearchDbPath, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils"; import { getEncodedProjectName } from "../task/worktree"; import * as git from "../utils/git"; import type { ASIData, ExperimentStatus, MetricDirection, NumericMetricMap } from "./types"; @@ -86,6 +86,7 @@ export interface UpdateSessionParams { metricUnit?: string; direction?: MetricDirection; branch?: string | null; + baselineCommit?: string | null; notes?: string; } @@ -278,6 +279,24 @@ export class AutoresearchStorage { return row ? rowToSession(row) : null; } + getActiveSessionForBranch(branch: string | null): SessionRow | null { + // Most-recent active session whose recorded branch matches the caller's branch. + // `branch === null` means "no git repo / no branch info" — treat null on both + // sides as a match. + if (branch === null) { + const stmt = this.#db.prepare( + "SELECT * FROM sessions WHERE closed_at IS NULL AND branch IS NULL ORDER BY id DESC LIMIT 1", + ); + const row = stmt.get(); + return row ? rowToSession(row) : null; + } + const stmt = this.#db.prepare( + "SELECT * FROM sessions WHERE closed_at IS NULL AND branch = ? ORDER BY id DESC LIMIT 1", + ); + const row = stmt.get(branch); + return row ? rowToSession(row) : null; + } + getSessionById(sessionId: number): SessionRow | null { const stmt = this.#db.prepare("SELECT * FROM sessions WHERE id = ?"); const row = stmt.get(sessionId); @@ -362,6 +381,10 @@ export class AutoresearchStorage { setClauses.push("branch = ?"); values.push(updates.branch); } + if (updates.baselineCommit !== undefined) { + setClauses.push("baseline_commit = ?"); + values.push(updates.baselineCommit); + } if (updates.notes !== undefined) { setClauses.push("notes = ?"); values.push(updates.notes); @@ -516,27 +539,41 @@ export class AutoresearchStorage { const storageCache = new Map(); export async function openAutoresearchStorage(cwd: string): Promise { - const override = process.env.OMP_AUTORESEARCH_DB_DIR; - const repoRoot = (await git.repo.root(cwd)) ?? cwd; - const encoded = getEncodedProjectName(repoRoot); - let dbPath: string; - let projectDir: string; - if (override) { - fs.mkdirSync(override, { recursive: true }); - dbPath = path.join(override, `${encoded}.db`); - projectDir = path.join(override, encoded); - } else { - dbPath = getAutoresearchDbPath(encoded); - projectDir = getAutoresearchProjectDir(encoded); - fs.mkdirSync(getAutoresearchDir(), { recursive: true }); - } + const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd); const cached = storageCache.get(dbPath); if (cached) return cached; + fs.mkdirSync(path.dirname(dbPath), { recursive: true }); const storage = new AutoresearchStorage(dbPath, projectDir); storageCache.set(dbPath, storage); return storage; } +export async function openAutoresearchStorageIfExists(cwd: string): Promise { + const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd); + const cached = storageCache.get(dbPath); + if (cached) return cached; + if (!fs.existsSync(dbPath)) return null; + const storage = new AutoresearchStorage(dbPath, projectDir); + storageCache.set(dbPath, storage); + return storage; +} + +async function resolveAutoresearchPaths(cwd: string): Promise<{ dbPath: string; projectDir: string }> { + const override = process.env.OMP_AUTORESEARCH_DB_DIR; + const repoRoot = (await git.repo.root(cwd)) ?? cwd; + const encoded = getEncodedProjectName(repoRoot); + if (override) { + return { + dbPath: path.join(override, `${encoded}.db`), + projectDir: path.join(override, encoded), + }; + } + return { + dbPath: getAutoresearchDbPath(encoded), + projectDir: getAutoresearchProjectDir(encoded), + }; +} + export function closeAllAutoresearchStorages(): void { for (const storage of storageCache.values()) { try { diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts index 6ad261254..9e09dcbe6 100644 --- a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -1,3 +1,4 @@ +import * as path from "node:path"; import { StringEnum } from "@oh-my-pi/pi-ai"; import { Text } from "@oh-my-pi/pi-tui"; import { Type } from "@sinclair/typebox"; @@ -5,11 +6,16 @@ import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; import * as git from "../../utils/git"; +import { parseWorkDirDirtyPaths } from "../git"; import { dedupeStrings, normalizePathSpec } from "../helpers"; import { buildExperimentState } from "../state"; import { openAutoresearchStorage, type SessionRow } from "../storage"; import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types"; +export const HARNESS_FILENAME = "autoresearch.sh"; +export const DEFAULT_HARNESS_COMMAND = `bash ${HARNESS_FILENAME}`; +const HARNESS_COMMIT_TITLE = "autoresearch: harness setup"; + const initExperimentSchema = Type.Object({ name: Type.String({ description: "Human-readable experiment name." }), goal: Type.Optional(Type.String({ description: "Free-form description of what this session optimizes." })), @@ -23,12 +29,6 @@ const initExperimentSchema = Type.Object({ direction: Type.Optional( StringEnum(["lower", "higher"], { description: "Whether lower or higher values are better. Defaults to lower." }), ), - preferred_command: Type.Optional( - Type.String({ - description: - "Preferred benchmark command for this segment. Advisory; run_experiment accepts any command but warns when the command differs.", - }), - ), secondary_metrics: Type.Optional( Type.Array(Type.String(), { description: "Names of secondary metrics tracked alongside the primary metric.", @@ -63,6 +63,8 @@ interface InitExperimentDetails { createdSession: boolean; bumpedSegment: boolean; abandonedRuns: number; + harnessCommitted: boolean; + baselineCommit: string | null; } export function createInitExperimentTool( @@ -72,7 +74,7 @@ export function createInitExperimentTool( name: "init_experiment", label: "Init Experiment", description: - "Initialize or reconfigure the autoresearch session. Pass `new_segment: true` to start a fresh baseline within an existing session.", + "Initialize or reconfigure the autoresearch session. On first call (Phase 1 → Phase 2 transition), requires `./autoresearch.sh` to exist and pending harness changes are auto-committed on an autoresearch branch. Pass `new_segment: true` to start a fresh baseline within an existing session.", parameters: initExperimentSchema, defaultInactive: true, async execute(_toolCallId, params, _signal, _onUpdate, ctx) { @@ -85,29 +87,63 @@ export function createInitExperimentTool( const offLimits = dedupeStrings((params.off_limits ?? []).map(normalizePathSpec)); const constraints = dedupeStrings(params.constraints ?? []); const secondaryMetrics = dedupeStrings(params.secondary_metrics ?? []); - const preferredCommand = params.preferred_command?.trim() || null; const goal = params.goal?.trim() || null; const maxIterations = params.max_iterations !== undefined && Number.isFinite(params.max_iterations) && params.max_iterations > 0 ? Math.floor(params.max_iterations) : null; const branch = (await git.branch.current(ctx.cwd)) ?? null; + const onAutoresearchBranch = branch?.startsWith("autoresearch/") ?? false; + + const existing = storage.getActiveSessionForBranch(branch); + const isNewSegmentInit = existing !== null && params.new_segment === true; + const requiresHarness = !existing || isNewSegmentInit; + + if (requiresHarness) { + const harnessExists = await Bun.file(path.join(ctx.cwd, HARNESS_FILENAME)).exists(); + if (!harnessExists) { + return { + content: [ + { + type: "text", + text: `Error: ./${HARNESS_FILENAME} does not exist. Phase 1 of autoresearch is harness setup — write \`./${HARNESS_FILENAME}\` so it exits 0 and prints \`METRIC =\`, validate it via \`bash ${HARNESS_FILENAME}\`, then call init_experiment again.`, + }, + ], + }; + } + } + + let harnessCommitted = false; + let commitWarning: string | null = null; + if (requiresHarness && onAutoresearchBranch) { + const dirty = await detectPendingChanges(ctx.cwd); + if (dirty) { + try { + await git.stage.files(ctx.cwd, []); + const message = buildHarnessCommitMessage(goal, params.name); + await git.commit(ctx.cwd, message); + harnessCommitted = true; + } catch (err) { + commitWarning = `Failed to auto-commit harness changes: ${err instanceof Error ? err.message : String(err)}. Recording baseline at current HEAD; discard may not preserve uncommitted harness files.`; + } + } + } + + const baselineCommit = await tryReadHeadSha(ctx.cwd); - const existing = storage.getActiveSession(); let session: SessionRow; let createdSession = false; let bumpedSegment = false; let abandonedRuns = 0; if (!existing) { - const baselineCommit = await tryReadHeadSha(ctx.cwd); session = storage.openSession({ name: params.name, goal, primaryMetric: params.primary_metric, metricUnit, direction, - preferredCommand, + preferredCommand: DEFAULT_HARNESS_COMMAND, branch, baselineCommit, maxIterations, @@ -119,9 +155,8 @@ export function createInitExperimentTool( createdSession = true; } else { abandonedRuns = storage.abandonPendingRuns(existing.id); - const updates = { + const updates: Parameters[1] = { goal, - preferredCommand, maxIterations, scopePaths, offLimits, @@ -132,8 +167,11 @@ export function createInitExperimentTool( direction, branch, }; + if (isNewSegmentInit) { + updates.baselineCommit = baselineCommit; + } let updated = storage.updateSession(existing.id, updates); - if (params.new_segment === true) { + if (isNewSegmentInit) { updated = storage.bumpSegment(existing.id); bumpedSegment = true; } @@ -159,6 +197,12 @@ export function createInitExperimentTool( if (abandonedRuns > 0) { lines.push(`Abandoned ${abandonedRuns} pending run${abandonedRuns === 1 ? "" : "s"} before reconfiguring.`); } + if (harnessCommitted && session.baselineCommit) { + lines.push(`Committed harness setup at ${session.baselineCommit.slice(0, 12)}.`); + } + if (commitWarning) { + lines.push(commitWarning); + } if (createdSession) { lines.push(`Started session #${session.id}: ${session.name}`); } else if (bumpedSegment) { @@ -169,9 +213,7 @@ export function createInitExperimentTool( lines.push( `Metric: ${session.primaryMetric} (${session.metricUnit || "unitless"}, ${session.direction} is better)`, ); - if (session.preferredCommand) { - lines.push(`Preferred command: ${session.preferredCommand}`); - } + lines.push(`Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`); if (session.scopePaths.length > 0) { lines.push(`Files in scope: ${session.scopePaths.join(", ")}`); } @@ -188,10 +230,17 @@ export function createInitExperimentTool( lines.push(`Baseline commit: ${session.baselineCommit.slice(0, 12)}`); } if (createdSession) { - lines.push("Run the baseline experiment now and log it."); + lines.push( + "Phase 2: iteration loop is active. Run the baseline experiment with `run_experiment` and log it.", + ); } else if (bumpedSegment) { lines.push("Run a fresh baseline for the new segment."); } + if (requiresHarness && !onAutoresearchBranch) { + lines.push( + "Note: not on a dedicated `autoresearch/*` branch — `log_experiment discard` will only revert run-modified files, not reset to baseline.", + ); + } return { content: [{ type: "text", text: lines.join("\n") }], @@ -200,6 +249,8 @@ export function createInitExperimentTool( createdSession, bumpedSegment, abandonedRuns, + harnessCommitted, + baselineCommit: session.baselineCommit, }, }; }, @@ -224,3 +275,23 @@ async function tryReadHeadSha(cwd: string): Promise { return null; } } + +async function detectPendingChanges(cwd: string): Promise { + try { + const statusText = await git.status(cwd, { porcelainV1: true, untrackedFiles: "all", z: true }); + const workDirPrefix = await git.show.prefix(cwd).catch(() => ""); + return parseWorkDirDirtyPaths(statusText, workDirPrefix).length > 0; + } catch { + return false; + } +} + +function buildHarnessCommitMessage(goal: string | null, name: string): string { + const lines = [HARNESS_COMMIT_TITLE, "", `Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`]; + if (goal) { + lines.push(`Goal: ${goal}`); + } else { + lines.push(`Session: ${name}`); + } + return lines.join("\n"); +} diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index b066c82d8..d21d4c0f1 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -7,7 +7,7 @@ import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; import * as git from "../../utils/git"; -import { computeRunModifiedPaths, getCurrentAutoresearchBranch } from "../git"; +import { computeRunModifiedPaths, getCurrentAutoresearchBranch, parseWorkDirDirtyPaths } from "../git"; import { ensureNumericMetricMap, formatNum, mergeAsi, pathMatchesSpec, sanitizeAsi } from "../helpers"; import { buildExperimentState, @@ -16,7 +16,7 @@ import { findBaselineSecondary, findBestKeptMetric, } from "../state"; -import { openAutoresearchStorage, type SessionRow } from "../storage"; +import { openAutoresearchStorageIfExists, type SessionRow } from "../storage"; import type { ASIData, AutoresearchToolFactoryOptions, @@ -81,14 +81,15 @@ export function createLogExperimentTool( parameters: logExperimentSchema, defaultInactive: true, async execute(_toolCallId, params, _signal, _onUpdate, ctx) { - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - if (!session) { + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + const currentBranch = (await git.branch.current(ctx.cwd)) ?? null; + const session = storage?.getActiveSessionForBranch(currentBranch) ?? null; + if (!storage || !session) { return { content: [ { type: "text", - text: "Error: no active autoresearch session. Call init_experiment first.", + text: "Error: no active autoresearch session for the current branch. Call init_experiment first.", }, ], }; @@ -113,8 +114,23 @@ export function createLogExperimentTool( const branchName = await getCurrentAutoresearchBranch(options.pi, ctx.cwd); const onAutoresearchBranch = branchName !== null; - const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths(ctx.cwd, pendingRun.preRunDirtyPaths); - const allModified = [...modifiedTracked, ...modifiedUntracked]; + let allModified: string[]; + if (onAutoresearchBranch) { + // On a dedicated autoresearch branch every iteration starts from a clean + // worktree (init_experiment baseline + previous keep commit / discard reset), + // so any currently-dirty path is the agent's iteration change. Off-branch we + // can't tell user dirt apart from agent edits, so we keep the (lossy) + // preRunDirtyPaths filter. + const statusText = await tryGitStatus(ctx.cwd); + const workDirPrefix = await tryGitPrefix(ctx.cwd); + allModified = parseWorkDirDirtyPaths(statusText, workDirPrefix); + } else { + const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths( + ctx.cwd, + pendingRun.preRunDirtyPaths, + ); + allModified = [...modifiedTracked, ...modifiedUntracked]; + } const scopeDeviations = computeScopeDeviations(allModified, session); const justification = params.justification?.trim() || null; @@ -165,7 +181,6 @@ export function createLogExperimentTool( ctx.cwd, pendingRun.preRunDirtyPaths, onAutoresearchBranch, - session.baselineCommit, ); if (revertResult.error) { return { @@ -346,14 +361,15 @@ async function revertFailedExperiment( cwd: string, preRunDirtyPaths: string[], onAutoresearchBranch: boolean, - baselineCommit: string | null, ): Promise { if (onAutoresearchBranch) { + // Discard reverts only the current iteration's uncommitted changes — never + // rewinds prior `keep` commits. Reset to HEAD so any kept improvements + // already on the branch survive. try { - const target = baselineCommit && baselineCommit.length > 0 ? baselineCommit : "HEAD"; - await git.reset(cwd, { hard: true, target }); + await git.reset(cwd, { hard: true, target: "HEAD" }); await git.clean(cwd); - return { note: `worktree reset to ${target.slice(0, 12)}` }; + return { note: "worktree reset to HEAD" }; } catch (err) { return { error: `git reset/clean failed: ${err instanceof Error ? err.message : String(err)}` }; } diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts index be2cfc16b..5041c611b 100644 --- a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -7,7 +7,7 @@ import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output"; -import { replaceTabs, shortenPath, truncateToWidth } from "../../tools/render-utils"; +import { replaceTabs, shortenPath } from "../../tools/render-utils"; import * as git from "../../utils/git"; import { parseWorkDirDirtyPaths } from "../git"; import { @@ -20,11 +20,11 @@ import { parseMetricLines, } from "../helpers"; import { buildExperimentState } from "../state"; -import { openAutoresearchStorage } from "../storage"; +import { openAutoresearchStorageIfExists } from "../storage"; import type { AutoresearchToolFactoryOptions, RunDetails, RunExperimentProgressDetails } from "../types"; +import { DEFAULT_HARNESS_COMMAND } from "./init-experiment"; const runExperimentSchema = Type.Object({ - command: Type.String({ description: "Shell command to run for this experiment." }), timeout_seconds: Type.Optional(Type.Number({ description: "Timeout in seconds. Defaults to 600." })), }); @@ -54,14 +54,15 @@ export function createRunExperimentTool( parameters: runExperimentSchema, defaultInactive: true, async execute(_toolCallId, params, signal, onUpdate, ctx) { - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - if (!session) { + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + const currentBranch = (await git.branch.current(ctx.cwd)) ?? null; + const session = storage?.getActiveSessionForBranch(currentBranch) ?? null; + if (!storage || !session) { return { content: [ { type: "text", - text: "Error: no active autoresearch session. Call init_experiment first.", + text: "Error: no active autoresearch session for the current branch. Call init_experiment first.", }, ], }; @@ -76,11 +77,7 @@ export function createRunExperimentTool( return pending.id; })(); - let commandWarning: string | null = null; - if (session.preferredCommand && params.command.trim() !== session.preferredCommand.trim()) { - commandWarning = `Note: command differs from preferred (\`${session.preferredCommand}\`). Re-init the experiment if the workload itself changed.`; - } - + const resolvedCommand = DEFAULT_HARNESS_COMMAND; const preRunStatus = await tryGitStatus(ctx.cwd); const workDirPrefix = await tryGitPrefix(ctx.cwd); const preRunDirtyPaths = parseWorkDirDirtyPaths(preRunStatus, workDirPrefix); @@ -89,7 +86,7 @@ export function createRunExperimentTool( const insertedRun = storage.insertRun({ sessionId: session.id, segment: session.currentSegment, - command: params.command, + command: resolvedCommand, logPath: "", // patched after we know the run id preRunDirtyPaths, startedAt, @@ -107,7 +104,7 @@ export function createRunExperimentTool( runtime.lastRunSummary = null; runtime.runningExperiment = { startedAt, - command: params.command, + command: resolvedCommand, runDirectory, runNumber: insertedRun.id, }; @@ -118,7 +115,7 @@ export function createRunExperimentTool( let execution: ProcessExecutionResult; try { execution = await executeProcess({ - command: ["bash", "-lc", params.command], + command: ["bash", "-lc", resolvedCommand], cwd: ctx.cwd, logPath: benchmarkLogPath, timeoutMs, @@ -178,7 +175,7 @@ export function createRunExperimentTool( runNumber: insertedRun.id, runDirectory, benchmarkLogPath, - command: params.command, + command: resolvedCommand, exitCode: execution.exitCode, durationSeconds, passed, @@ -191,14 +188,13 @@ export function createRunExperimentTool( metricName: session.primaryMetric, metricUnit: session.metricUnit, preRunDirtyPaths, - commandWarning, abandonedPriorRun, truncation: llmTruncation.truncated ? llmTruncation : undefined, fullOutputPath: execution.logPath, }; runtime.lastRunSummary = { - command: params.command, + command: resolvedCommand, durationSeconds, parsedAsi, parsedMetrics, @@ -222,7 +218,6 @@ export function createRunExperimentTool( options.dashboard.requestRender(); const headerLines: string[] = []; - if (commandWarning) headerLines.push(commandWarning); if (abandonedPriorRun !== null) { headerLines.push(`Note: abandoned prior pending run #${abandonedPriorRun} before starting this run.`); } @@ -238,10 +233,9 @@ export function createRunExperimentTool( details: resultDetails, }; }, - renderCall(args, _options, theme): Text { - const commandPreview = truncateToWidth(replaceTabs(args.command), 100); + renderCall(_args, _options, theme): Text { return new Text( - `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", commandPreview)}`, + `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", DEFAULT_HARNESS_COMMAND)}`, 0, 0, ); diff --git a/packages/coding-agent/src/autoresearch/tools/update-notes.ts b/packages/coding-agent/src/autoresearch/tools/update-notes.ts index bde8f6d93..bf42dd9a9 100644 --- a/packages/coding-agent/src/autoresearch/tools/update-notes.ts +++ b/packages/coding-agent/src/autoresearch/tools/update-notes.ts @@ -3,8 +3,9 @@ import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; +import * as git from "../../utils/git"; import { buildExperimentState } from "../state"; -import { openAutoresearchStorage } from "../storage"; +import { openAutoresearchStorageIfExists } from "../storage"; import type { AutoresearchToolFactoryOptions } from "../types"; const updateNotesSchema = Type.Object({ @@ -34,14 +35,15 @@ export function createUpdateNotesTool( parameters: updateNotesSchema, defaultInactive: true, async execute(_toolCallId, params, _signal, _onUpdate, ctx) { - const storage = await openAutoresearchStorage(ctx.cwd); - const session = storage.getActiveSession(); - if (!session) { + const storage = await openAutoresearchStorageIfExists(ctx.cwd); + const currentBranch = (await git.branch.current(ctx.cwd)) ?? null; + const session = storage?.getActiveSessionForBranch(currentBranch) ?? null; + if (!storage || !session) { return { content: [ { type: "text", - text: "Error: no active autoresearch session. Call init_experiment first.", + text: "Error: no active autoresearch session for the current branch. Call init_experiment first.", }, ], }; diff --git a/packages/coding-agent/src/autoresearch/types.ts b/packages/coding-agent/src/autoresearch/types.ts index be6d51ea0..f442166fe 100644 --- a/packages/coding-agent/src/autoresearch/types.ts +++ b/packages/coding-agent/src/autoresearch/types.ts @@ -51,7 +51,6 @@ export interface ExperimentState { currentSegment: number; maxExperiments: number | null; confidence: number | null; - benchmarkCommand: string | null; scopePaths: string[]; offLimits: string[]; constraints: string[]; @@ -86,7 +85,6 @@ export interface RunDetails { metricName: string; metricUnit: string; preRunDirtyPaths: string[]; - commandWarning: string | null; abandonedPriorRun: number | null; truncation?: TruncationResult; fullOutputPath?: string; diff --git a/packages/coding-agent/test/autoresearch-tools.test.ts b/packages/coding-agent/test/autoresearch-tools.test.ts index 8d05ffd11..3b2c173f0 100644 --- a/packages/coding-agent/test/autoresearch-tools.test.ts +++ b/packages/coding-agent/test/autoresearch-tools.test.ts @@ -84,6 +84,10 @@ async function checkoutBranch(dir: string, name: string): Promise { await $`git checkout -b ${name}`.cwd(dir).quiet(); } +async function writeHarnessStub(dir: string, body = "echo METRIC m=1"): Promise { + await Bun.write(path.join(dir, "autoresearch.sh"), `#!/usr/bin/env bash\n${body}\n`); +} + describe("init_experiment", () => { let dbOverride: string; @@ -99,6 +103,7 @@ describe("init_experiment", () => { it("opens a new session and persists scope and metric metadata", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const tool = createInitExperimentTool({ dashboard: dashboardStub(), @@ -114,7 +119,6 @@ describe("init_experiment", () => { primary_metric: "runtime_ms", metric_unit: "ms", direction: "lower", - preferred_command: "bun bench", scope_paths: ["src", "src/foo"], off_limits: ["test"], secondary_metrics: ["memory_mb"], @@ -141,6 +145,7 @@ describe("init_experiment", () => { it("updates fields without bumping segment when no new_segment flag is passed", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const tool = createInitExperimentTool({ dashboard: dashboardStub(), @@ -171,6 +176,7 @@ describe("init_experiment", () => { it("bumps segment when new_segment is true on a re-init", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const tool = createInitExperimentTool({ dashboard: dashboardStub(), @@ -188,6 +194,78 @@ describe("init_experiment", () => { expect(result.details?.bumpedSegment).toBe(true); expect(result.details?.state.currentSegment).toBe(1); }); + + it("rejects when autoresearch.sh is missing on first init", async () => { + const dir = makeTempDir(); + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: createPiHarness().api, + }); + const result = await tool.execute( + "call-1", + { name: "x", primary_metric: "m" }, + undefined, + undefined, + createCtx(dir), + ); + expect(firstTextBlockText(result.content)).toContain("autoresearch.sh"); + const storage = await openAutoresearchStorage(dir); + expect(storage.getActiveSession()).toBeNull(); + }); + + it("auto-commits pending harness changes on an autoresearch branch", async () => { + const dir = makeTempDir(); + const { baselineCommit: initialBaseline } = await initGitRepo(dir); + await checkoutBranch(dir, "autoresearch/setup-test"); + await writeHarnessStub(dir); + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: createPiHarness().api, + }); + const result = await tool.execute( + "call-1", + { name: "x", primary_metric: "m", goal: "speed" }, + undefined, + undefined, + createCtx(dir), + ); + expect(result.details?.harnessCommitted).toBe(true); + const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); + expect(newHead).not.toBe(initialBaseline); + expect(result.details?.baselineCommit).toBe(newHead); + const status = (await $`git status --porcelain`.cwd(dir).text()).trim(); + expect(status).toBe(""); + const message = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim(); + expect(message).toContain("autoresearch: harness setup"); + }); + + it("does not auto-commit when not on an autoresearch branch", async () => { + const dir = makeTempDir(); + const { baselineCommit: initialBaseline } = await initGitRepo(dir); + await writeHarnessStub(dir); + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: createPiHarness().api, + }); + const result = await tool.execute( + "call-1", + { name: "x", primary_metric: "m" }, + undefined, + undefined, + createCtx(dir), + ); + expect(result.details?.harnessCommitted).toBe(false); + const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); + expect(newHead).toBe(initialBaseline); + // Harness file is still in the worktree, untracked. + expect(fs.existsSync(path.join(dir, "autoresearch.sh"))).toBe(true); + }); }); describe("run_experiment", () => { @@ -211,12 +289,13 @@ describe("run_experiment", () => { getRuntime: () => runtime, pi: createPiHarness().api, }); - const result = await run.execute("call-1", { command: "echo hi" }, undefined, undefined, createCtx(dir)); + const result = await run.execute("call-1", {}, undefined, undefined, createCtx(dir)); expect(firstTextBlockText(result.content)).toContain("no active autoresearch session"); }); it("accepts arbitrary commands, parses METRIC/ASI, and stores a run", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir, "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline"); const runtime = createSessionRuntime(); const init = createInitExperimentTool({ dashboard: dashboardStub(), @@ -235,16 +314,7 @@ describe("run_experiment", () => { getRuntime: () => runtime, pi: createPiHarness().api, }); - const result = await run.execute( - "r", - { - command: "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline", - timeout_seconds: 5, - }, - undefined, - undefined, - createCtx(dir), - ); + const result = await run.execute("r", { timeout_seconds: 5 }, undefined, undefined, createCtx(dir)); const details = result.details as RunDetails; expect(details.parsedPrimary).toBe(42); expect(details.parsedMetrics).toMatchObject({ runtime_ms: 42, memory_mb: 12 }); @@ -262,6 +332,7 @@ describe("run_experiment", () => { it("abandons a prior pending run instead of blocking", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const initTool = createInitExperimentTool({ dashboard: dashboardStub(), @@ -274,37 +345,32 @@ describe("run_experiment", () => { getRuntime: () => runtime, pi: createPiHarness().api, }); - await run.execute("r1", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir)); - const result = await run.execute("r2", { command: "echo METRIC m=2" }, undefined, undefined, createCtx(dir)); + await run.execute("r1", {}, undefined, undefined, createCtx(dir)); + const result = await run.execute("r2", {}, undefined, undefined, createCtx(dir)); const details = result.details as RunDetails; expect(details.abandonedPriorRun).not.toBeNull(); expect(details.runNumber).not.toBe(details.abandonedPriorRun); }); - it("warns when command differs from the preferred command", async () => { + it("runs ./autoresearch.sh and parses METRIC/ASI from its output", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir, "echo METRIC m=99"); const runtime = createSessionRuntime(); const init = createInitExperimentTool({ dashboard: dashboardStub(), getRuntime: () => runtime, pi: createPiHarness().api, }); - await init.execute( - "i", - { name: "x", primary_metric: "m", preferred_command: "echo preferred" }, - undefined, - undefined, - createCtx(dir), - ); + await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir)); const run = createRunExperimentTool({ dashboard: dashboardStub(), getRuntime: () => runtime, pi: createPiHarness().api, }); - const result = await run.execute("r", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir)); + const result = await run.execute("r", {}, undefined, undefined, createCtx(dir)); const details = result.details as RunDetails; - expect(details.commandWarning).toContain("preferred"); - expect(firstTextBlockText(result.content)).toContain("preferred"); + expect(details.command).toBe("bash autoresearch.sh"); + expect(details.parsedPrimary).toBe(99); }); }); @@ -322,6 +388,7 @@ describe("log_experiment", () => { }); async function setupRun(dir: string, runtime = createSessionRuntime()) { + await writeHarnessStub(dir, "echo METRIC runtime_ms=10"); const harness = createPiHarness(); const init = createInitExperimentTool({ dashboard: dashboardStub(), @@ -346,7 +413,7 @@ describe("log_experiment", () => { getRuntime: () => runtime, pi: harness.api, }); - await run.execute("r", { command: "echo METRIC runtime_ms=10" }, undefined, undefined, createCtx(dir)); + await run.execute("r", {}, undefined, undefined, createCtx(dir)); const log = createLogExperimentTool({ dashboard: dashboardStub(), getRuntime: () => runtime, @@ -357,6 +424,7 @@ describe("log_experiment", () => { it("rejects when no pending run exists", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const harness = createPiHarness(); const init = createInitExperimentTool({ @@ -480,7 +548,7 @@ describe("log_experiment", () => { getRuntime: () => runtime, pi: harness.api, }); - await run.execute("r2", { command: "echo METRIC runtime_ms=8" }, undefined, undefined, createCtx(dir)); + await run.execute("r2", {}, undefined, undefined, createCtx(dir)); const log2 = createLogExperimentTool({ dashboard: dashboardStub(), getRuntime: () => runtime, @@ -512,6 +580,7 @@ describe("log_experiment", () => { it("on a non-autoresearch branch, discard reverts only run-modified files", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); await initGitRepo(dir); // Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths. fs.mkdirSync(path.join(dir, "src"), { recursive: true }); @@ -539,7 +608,7 @@ describe("log_experiment", () => { }); // Pre-existing untracked file (will not be touched by revert because it was dirty before run) await Bun.write(path.join(dir, "preexisting.txt"), "leave me\n"); - await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir)); + await run.execute("r", {}, undefined, undefined, createCtx(dir)); // Simulate a run-introduced change await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 2;\n"); await Bun.write(path.join(dir, "src", "new.ts"), "export const NEW = true;\n"); @@ -564,9 +633,13 @@ describe("log_experiment", () => { expect(fs.readFileSync(path.join(dir, "src", "edit-me.ts"), "utf8")).toBe("export const v = 1;\n"); }); - it("on an autoresearch branch, discard resets the worktree to baseline_commit", async () => { + it("on an autoresearch branch, discard reverts uncommitted changes but preserves prior commits", async () => { const dir = makeTempDir(); - const { baselineCommit } = await initGitRepo(dir); + await initGitRepo(dir); + // Commit the harness on main so it is part of the autoresearch branch's baseline. + await writeHarnessStub(dir); + await $`git add -A`.cwd(dir).quiet(); + await $`git commit -m harness`.cwd(dir).quiet(); await checkoutBranch(dir, "autoresearch/test-20260501"); const runtime = createSessionRuntime(); const harness = createPiHarness(); @@ -576,16 +649,21 @@ describe("log_experiment", () => { pi: harness.api, }); await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir)); + // Simulate a previously kept iteration by committing it directly on the branch. + await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n"); + await $`git add -A`.cwd(dir).quiet(); + await $`git commit -m "kept iteration"`.cwd(dir).quiet(); + const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); + const run = createRunExperimentTool({ dashboard: dashboardStub(), getRuntime: () => runtime, pi: harness.api, }); - await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir)); - // Modify and commit a file on the autoresearch branch — discard should reset HEAD back to baseline. - await Bun.write(path.join(dir, "src", "stub.ts"), "export const v = 1;\n"); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m wip`.cwd(dir).quiet(); + await run.execute("r", {}, undefined, undefined, createCtx(dir)); + // Current iteration's uncommitted edits. + await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 999;\n"); + await Bun.write(path.join(dir, "scratch.ts"), "// junk\n"); const log = createLogExperimentTool({ dashboard: dashboardStub(), @@ -599,9 +677,117 @@ describe("log_experiment", () => { undefined, createCtx(dir), ); - const headSha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); - expect(headSha).toBe(baselineCommit); - expect(fs.existsSync(path.join(dir, "src", "stub.ts"))).toBe(false); + const headAfter = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); + // Prior commits survive — discard does not rewind history. + expect(headAfter).toBe(headBeforeDiscard); + // Uncommitted iteration changes are gone. + expect(fs.readFileSync(path.join(dir, "src", "kept.ts"), "utf8")).toBe("export const v = 1;\n"); + expect(fs.existsSync(path.join(dir, "scratch.ts"))).toBe(false); + const status = (await $`git status --porcelain`.cwd(dir).text()).trim(); + expect(status).toBe(""); + }); + + it("on an autoresearch branch, keep commits files that were dirty before run_experiment", async () => { + const dir = makeTempDir(); + await initGitRepo(dir); + await writeHarnessStub(dir); + await $`git add -A`.cwd(dir).quiet(); + await $`git commit -m harness`.cwd(dir).quiet(); + // Seed a tracked file that the agent will edit during the iteration. + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n"); + await $`git add -A`.cwd(dir).quiet(); + await $`git commit -m seed`.cwd(dir).quiet(); + await checkoutBranch(dir, "autoresearch/keep-test"); + const runtime = createSessionRuntime(); + const harness = createPiHarness(); + const init = createInitExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + await init.execute( + "i", + { name: "x", primary_metric: "m", scope_paths: ["src"] }, + undefined, + undefined, + createCtx(dir), + ); + // Agent edits BEFORE running the benchmark — the iteration's diff is dirty + // at run_experiment time. + await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 2;\n"); + const run = createRunExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + await run.execute("r", {}, undefined, undefined, createCtx(dir)); + + const log = createLogExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + const result = await log.execute( + "l", + { metric: 42, status: "keep", description: "improvement" }, + undefined, + undefined, + createCtx(dir), + ); + const details = result.details as LogDetails; + expect(details.experiment.modifiedPaths).toContain("src/store.ts"); + const status = (await $`git status --porcelain`.cwd(dir).text()).trim(); + expect(status).toBe(""); + const lastMsg = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim(); + expect(lastMsg).toContain("improvement"); + }); + + it("flags off-scope dirty files even when they were dirty before run_experiment", async () => { + const dir = makeTempDir(); + await initGitRepo(dir); + await writeHarnessStub(dir); + await $`git add -A`.cwd(dir).quiet(); + await $`git commit -m harness`.cwd(dir).quiet(); + await checkoutBranch(dir, "autoresearch/scope-test"); + const runtime = createSessionRuntime(); + const harness = createPiHarness(); + const init = createInitExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + await init.execute( + "i", + { name: "x", primary_metric: "m", scope_paths: ["src"], off_limits: ["forbidden"] }, + undefined, + undefined, + createCtx(dir), + ); + // Off-scope edit BEFORE run_experiment. + fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true }); + await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n"); + const run = createRunExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + await run.execute("r", {}, undefined, undefined, createCtx(dir)); + + const log = createLogExperimentTool({ + dashboard: dashboardStub(), + getRuntime: () => runtime, + pi: harness.api, + }); + const result = await log.execute( + "l", + { metric: 42, status: "keep", description: "off-scope" }, + undefined, + undefined, + createCtx(dir), + ); + const details = result.details as LogDetails; + expect(details.scopeDeviations).toContain("forbidden/x.ts"); }); }); @@ -620,6 +806,7 @@ describe("update_notes", () => { it("replaces session notes and refreshes runtime state", async () => { const dir = makeTempDir(); + await writeHarnessStub(dir); const runtime = createSessionRuntime(); const harness = createPiHarness(); const init = createInitExperimentTool({