feat(coding-agent/autoresearch): added branch-specific state restore

- Added branch-aware session loading so autoresearch state only rehydrates for current branch.
- Replaced user-specified experiment commands with fixed `bash autoresearch.sh` execution flow.
- Enforced safer setup checks, including missing `autoresearch.sh` and uncommitted-worktree errors.
- Added branch-specific storage helpers, baseline-commit persistence, and expanded tests for dirty-path cases.
This commit is contained in:
can1357
2026-05-02 03:31:27 +02:00
parent 0698ad3ba3
commit 31ae1b96fd
12 changed files with 556 additions and 156 deletions
@@ -67,10 +67,8 @@ export async function ensureAutoresearchBranch(
if (dirtyPaths.length > 0) {
const preview = formatDirtyPaths(dirtyPaths);
return {
ok: true,
branchName: null,
created: false,
warning: `Worktree is dirty (${preview}). Continuing on the current branch; auto-commit and full-tree reset are disabled until you commit/stash these changes.`,
ok: false,
error: `Worktree is dirty (${preview}). Commit or stash these changes before starting autoresearch — a fresh autoresearch/* branch needs a clean baseline.`,
};
}
+90 -29
View File
@@ -9,6 +9,7 @@ import { createDashboardController } from "./dashboard";
import { ensureAutoresearchBranch } from "./git";
import { formatNum } from "./helpers";
import promptTemplate from "./prompt.md" with { type: "text" };
import setupPromptTemplate from "./prompt-setup.md" with { type: "text" };
import resumeMessageTemplate from "./resume-message.md" with { type: "text" };
import {
buildExperimentState,
@@ -20,7 +21,7 @@ import {
findBestKeptMetric,
reconstructControlState,
} from "./state";
import { openAutoresearchStorage, type RunRow } from "./storage";
import { openAutoresearchStorage, openAutoresearchStorageIfExists, type RunRow, type SessionRow } from "./storage";
import { createInitExperimentTool } from "./tools/init-experiment";
import { createLogExperimentTool } from "./tools/log-experiment";
import { createRunExperimentTool } from "./tools/run-experiment";
@@ -36,21 +37,49 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
const getSessionKey = (ctx: ExtensionContext): string => ctx.sessionManager.getSessionId();
const getRuntime = (ctx: ExtensionContext): AutoresearchRuntime => runtimeStore.ensure(getSessionKey(ctx));
const loadActiveSession = async (
ctx: ExtensionContext,
): Promise<{ session: SessionRow | null; currentBranch: string | null }> => {
const currentBranch = await tryReadBranch(ctx.cwd);
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
if (!storage) return { session: null, currentBranch };
const session = storage.getActiveSessionForBranch(currentBranch);
return { session, currentBranch };
};
const rehydrate = async (ctx: ExtensionContext): Promise<void> => {
const runtime = getRuntime(ctx);
const control = reconstructControlState(ctx.sessionManager.getBranch());
runtime.goal = control.goal;
runtime.autoresearchMode = control.autoresearchMode;
runtime.autoResumeArmed = false;
runtime.lastAutoResumePendingRunNumber = null;
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
if (session) {
const loggedRuns = storage.listLoggedRuns(session.id);
runtime.state = buildExperimentState(session, loggedRuns);
runtime.goal = runtime.goal ?? session.goal;
runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id));
// Skip storage entirely if autoresearch was never activated in this conversation.
// This is the common case: every project gets a session_start event but most
// never touch autoresearch, so we must not create a SQLite file just to look.
const everActivated = control.lastMode !== null;
const { session, currentBranch } = everActivated
? await loadActiveSession(ctx)
: { session: null, currentBranch: null };
// Mode is effective only when the recorded session matches the current git
// branch. When the user switches off the autoresearch branch the widget hides
// and the experiment tools detach, but the session entries are preserved so
// switching back resumes seamlessly.
const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch;
runtime.autoresearchMode = control.autoresearchMode && onActiveBranch;
if (session && onActiveBranch) {
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
if (storage) {
const loggedRuns = storage.listLoggedRuns(session.id);
runtime.state = buildExperimentState(session, loggedRuns);
runtime.goal = runtime.goal ?? session.goal;
runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id));
} else {
runtime.state = createExperimentState();
runtime.lastRunSummary = null;
}
} else {
runtime.state = createExperimentState();
runtime.lastRunSummary = null;
@@ -151,8 +180,12 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
ctx.ui.notify(branchResult.warning, "warning");
}
const storage = await openAutoresearchStorage(ctx.cwd);
const existingSession = storage.getActiveSession();
// Look up an existing session for the branch we just landed on. A session
// recorded under a different autoresearch/* branch is intentionally ignored
// — `/autoresearch` on a fresh branch starts a fresh session. Only open the
// DB if it already exists; the empty-state path must not create one.
const existingStorage = await openAutoresearchStorageIfExists(ctx.cwd);
const existingSession = existingStorage?.getActiveSessionForBranch(branchResult.branchName) ?? null;
const resumeContext = trimmed;
const branchStatusLine = branchResult.branchName
? branchResult.created
@@ -160,13 +193,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
: `Using dedicated git branch \`${branchResult.branchName}\`.`
: "Continuing on the current branch — no autoresearch branch was created.";
if (existingSession) {
if (goalArg) storage.updateSession(existingSession.id, { goal: goalArg });
if (existingSession && existingStorage) {
if (goalArg) existingStorage.updateSession(existingSession.id, { goal: goalArg });
if (branchResult.branchName) {
storage.updateSession(existingSession.id, { branch: branchResult.branchName });
existingStorage.updateSession(existingSession.id, { branch: branchResult.branchName });
}
const refreshed = storage.getSessionById(existingSession.id) ?? existingSession;
runtime.state = buildExperimentState(refreshed, storage.listLoggedRuns(refreshed.id));
const refreshed = existingStorage.getSessionById(existingSession.id) ?? existingSession;
runtime.state = buildExperimentState(refreshed, existingStorage.listLoggedRuns(refreshed.id));
runtime.goal = refreshed.goal ?? goalArg;
setMode(ctx, true, runtime.goal, "on");
dashboard.updateWidget(ctx, runtime);
@@ -231,9 +264,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
runtime.autoResumeArmed = false;
return;
}
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
const pendingRow = session ? storage.getPendingRun(session.id) : null;
const { session } = await loadActiveSession(ctx);
const storage = session ? await openAutoresearchStorageIfExists(ctx.cwd) : null;
const pendingRow = session && storage ? storage.getPendingRun(session.id) : null;
const pendingRun = pendingRunSummaryFromRow(pendingRow);
runtime.lastRunSummary = pendingRun;
runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration;
@@ -261,12 +294,27 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
api.on("before_agent_start", async (event, ctx) => {
const runtime = getRuntime(ctx);
if (!runtime.autoresearchMode) return;
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
if (session) {
// Re-check git branch on every agent start. If the user manually switched
// off the autoresearch/* branch between turns, we silently drop autoresearch
// from this turn — the widget hides, the experiment tools detach, and we do
// not inject the autoresearch system prompt.
const { session, currentBranch } = await loadActiveSession(ctx);
const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch;
if (!onActiveBranch) {
runtime.autoresearchMode = false;
runtime.state = createExperimentState();
runtime.lastRunSummary = null;
runtime.runningExperiment = null;
dashboard.updateWidget(ctx, runtime);
const experimentTools = new Set(EXPERIMENT_TOOL_NAMES);
await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name)));
return;
}
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
if (session && storage) {
runtime.state = buildExperimentState(session, storage.listLoggedRuns(session.id));
}
const pendingRow = session ? storage.getPendingRun(session.id) : null;
const pendingRow = session && storage ? storage.getPendingRun(session.id) : null;
const pendingRun = pendingRunSummaryFromRow(pendingRow);
runtime.lastRunSummary = pendingRun;
runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration;
@@ -301,9 +349,25 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
run_number: r.runNumber,
paths: r.scopeDeviations.join(", "),
}));
const lastCommand = pendingRun?.command ?? null;
const showCommandWarning =
Boolean(state.benchmarkCommand) && lastCommand !== null && lastCommand !== state.benchmarkCommand;
if (!session) {
const currentBranch = await tryReadBranch(ctx.cwd);
const onAutoresearchBranch = currentBranch?.startsWith("autoresearch/") ?? false;
const baselineWarning = onAutoresearchBranch
? null
: "Heads up: you are not on a dedicated `autoresearch/*` branch. `log_experiment discard` will only revert run-modified files, not reset to baseline — so harness files written before `init_experiment` may not survive a discard. Clean the worktree and re-run `/autoresearch` if you want full revert safety.";
return {
systemPrompt: prompt.render(setupPromptTemplate, {
base_system_prompt: event.systemPrompt,
has_goal: goal.trim().length > 0,
goal,
working_dir: ctx.cwd,
has_branch: Boolean(currentBranch),
branch: currentBranch ?? "",
has_baseline_warning: baselineWarning !== null,
baseline_warning: baselineWarning ?? "",
}),
};
}
return {
systemPrompt: prompt.render(promptTemplate, {
base_system_prompt: event.systemPrompt,
@@ -339,9 +403,6 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined
? formatNum(pendingRun.parsedPrimary, state.metricUnit)
: null,
has_preferred_command_warning: showCommandWarning,
preferred_command: state.benchmarkCommand ?? "",
last_command: lastCommand ?? "",
}),
};
});
@@ -0,0 +1,43 @@
{{base_system_prompt}}
## Autoresearch Mode — Phase 1: Harness Setup
Autoresearch mode is active and there is no session yet. Your job in this turn is to **build the benchmark harness**, not to optimise anything. Optimisation starts only after you call `init_experiment`.
{{#if has_goal}}
Primary goal (for context — implement the harness so it can measure this):
{{goal}}
{{else}}
There is no goal recorded yet. Infer what to optimise from the latest user message and design the harness to measure that. Capture the goal when you call `init_experiment`.
{{/if}}
Working directory: `{{working_dir}}`
{{#if has_branch}}Active branch: `{{branch}}`{{/if}}
{{#if has_baseline_warning}}
{{baseline_warning}}
{{/if}}
### What you must produce
Write `./autoresearch.sh` at the working directory. It is the canonical benchmark entrypoint and must:
- exit 0 on success and non-zero on failure;
- print the primary metric as a single line `METRIC <name>=<value>`;
- print any secondary metrics as additional `METRIC <name>=<value>` lines;
- run the same workload deterministically every time (no live network, no time-of-day dependencies, fixed seeds where applicable).
You **may** edit anything else needed to make `autoresearch.sh` work — benchmark binaries, `Cargo.toml`, `package.json`, helper scripts, fixtures. All those edits are part of the harness baseline and will be committed for you when you call `init_experiment` on an autoresearch branch.
### Steps
1. Inspect the target. Read source, identify what to measure, decide on the workload.
2. Write `autoresearch.sh` plus any supporting files (benchmark binaries, fixtures, etc.).
3. Validate it: invoke `bash autoresearch.sh` through the regular `bash` tool. Confirm it exits 0 and emits at least one `METRIC` line. Iterate on the harness until it does.
4. Call `init_experiment` with the goal, primary metric (matching the `METRIC` name), and scope. This snapshots the worktree as the baseline and starts Phase 2 (the iteration loop).
### Rules
- Do **not** call `run_experiment`, `log_experiment`, or `update_notes` yet. They will error with "no active autoresearch session" until `init_experiment` runs.
- Do **not** treat a compile-only check as a benchmark. The harness must actually execute the workload and emit `METRIC`.
- Do **not** create `autoresearch.md`, `autoresearch.checks.sh`, `autoresearch.program.md`, `autoresearch.ideas.md`, `autoresearch.jsonl`, `.autoresearch/`, or `autoresearch.config.json`. Session state is tracked for you.
@@ -11,7 +11,7 @@ Primary goal:
There is no goal recorded for this session yet. Infer what to optimize from the latest user message and the conversation; capture the goal in your notes (`update_notes`) once it is clear.
{{/if}}
Session state and run artifacts are managed for you. Do not create `autoresearch.md`, `autoresearch.sh`, or `.autoresearch/` in this repo.
Session state and run artifacts are managed for you. The benchmark entrypoint is `bash autoresearch.sh` (committed during Phase 1). Do not edit `autoresearch.sh` mid-segment unless you intentionally bump segment via `init_experiment new_segment: true`. Do not create `autoresearch.md` or `.autoresearch/` in this repo.
Working directory: `{{working_dir}}`
{{#if has_branch}}Active branch: `{{branch}}`{{/if}}
@@ -21,13 +21,13 @@ You are running an autonomous experiment loop. Keep iterating until the user int
### Available tools
- `init_experiment` — open or reconfigure the session. Pass `new_segment: true` to start a fresh baseline within the current session.
- `run_experiment` — run any benchmark command. Pass the actual command; output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the command are parsed back to you.
- `run_experiment` — run the benchmark (`bash autoresearch.sh`). Output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the harness are parsed back to you. The command is fixed; if you need a different workload, edit `autoresearch.sh` and bump segment via `init_experiment new_segment: true`.
- `log_experiment` — record the result. On `keep`, modified files are committed for you; on `discard`/`crash`/`checks_failed`, the worktree is reverted. Pass `flag_runs` to mark earlier runs as suspect; flagged runs are excluded from baseline and best-metric math.
- `update_notes` — replace the durable session playbook (`body`) or append to the ideas backlog (`append_idea`). The notes are injected into your system prompt every iteration.
### Operating protocol
1. Understand the target before touching code: read source, identify the bottleneck, verify prerequisites and benchmark inputs.
2. Capture goal, benchmark command, primary metric, scope, and constraints in `init_experiment`. Update them later via another `init_experiment` call (no segment bump) or via `update_notes`.
2. Update goal, scope, or constraints via another `init_experiment` call (no segment bump) or `update_notes`. Bump segment when you intentionally change `autoresearch.sh`.
3. Establish a baseline first.
4. Iterate: change code, run `run_experiment`, log honestly with `log_experiment`. One coherent experiment per iteration.
5. Keep the primary metric as the decision maker:
@@ -95,11 +95,6 @@ An unlogged run is waiting:
Finish the `log_experiment` step before starting another benchmark.
{{/if}}
{{#if has_preferred_command_warning}}
### Preferred command
Last `run_experiment` used `{{last_command}}`. Preferred command for this segment is `{{preferred_command}}`. If the workload changed intentionally, call `init_experiment` to update the preferred command (or pass `new_segment: true` to start a fresh baseline).
{{/if}}
### Guardrails
- Do not game the benchmark.
@@ -26,7 +26,6 @@ export function createExperimentState(): ExperimentState {
currentSegment: 0,
maxExperiments: null,
confidence: null,
benchmarkCommand: null,
scopePaths: [],
offLimits: [],
constraints: [],
@@ -177,7 +176,6 @@ export function buildExperimentState(session: SessionRow, loggedRuns: RunRow[]):
state.metricName = session.primaryMetric;
state.metricUnit = session.metricUnit;
state.bestDirection = session.direction;
state.benchmarkCommand = session.preferredCommand;
state.scopePaths = [...session.scopePaths];
state.offLimits = [...session.offLimits];
state.constraints = [...session.constraints];
@@ -1,7 +1,7 @@
import { Database, type SQLQueryBindings } from "bun:sqlite";
import * as fs from "node:fs";
import * as path from "node:path";
import { getAutoresearchDbPath, getAutoresearchDir, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils";
import { getAutoresearchDbPath, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils";
import { getEncodedProjectName } from "../task/worktree";
import * as git from "../utils/git";
import type { ASIData, ExperimentStatus, MetricDirection, NumericMetricMap } from "./types";
@@ -86,6 +86,7 @@ export interface UpdateSessionParams {
metricUnit?: string;
direction?: MetricDirection;
branch?: string | null;
baselineCommit?: string | null;
notes?: string;
}
@@ -278,6 +279,24 @@ export class AutoresearchStorage {
return row ? rowToSession(row) : null;
}
getActiveSessionForBranch(branch: string | null): SessionRow | null {
// Most-recent active session whose recorded branch matches the caller's branch.
// `branch === null` means "no git repo / no branch info" — treat null on both
// sides as a match.
if (branch === null) {
const stmt = this.#db.prepare<SessionDbRow, []>(
"SELECT * FROM sessions WHERE closed_at IS NULL AND branch IS NULL ORDER BY id DESC LIMIT 1",
);
const row = stmt.get();
return row ? rowToSession(row) : null;
}
const stmt = this.#db.prepare<SessionDbRow, [string]>(
"SELECT * FROM sessions WHERE closed_at IS NULL AND branch = ? ORDER BY id DESC LIMIT 1",
);
const row = stmt.get(branch);
return row ? rowToSession(row) : null;
}
getSessionById(sessionId: number): SessionRow | null {
const stmt = this.#db.prepare<SessionDbRow, [number]>("SELECT * FROM sessions WHERE id = ?");
const row = stmt.get(sessionId);
@@ -362,6 +381,10 @@ export class AutoresearchStorage {
setClauses.push("branch = ?");
values.push(updates.branch);
}
if (updates.baselineCommit !== undefined) {
setClauses.push("baseline_commit = ?");
values.push(updates.baselineCommit);
}
if (updates.notes !== undefined) {
setClauses.push("notes = ?");
values.push(updates.notes);
@@ -516,27 +539,41 @@ export class AutoresearchStorage {
const storageCache = new Map<string, AutoresearchStorage>();
export async function openAutoresearchStorage(cwd: string): Promise<AutoresearchStorage> {
const override = process.env.OMP_AUTORESEARCH_DB_DIR;
const repoRoot = (await git.repo.root(cwd)) ?? cwd;
const encoded = getEncodedProjectName(repoRoot);
let dbPath: string;
let projectDir: string;
if (override) {
fs.mkdirSync(override, { recursive: true });
dbPath = path.join(override, `${encoded}.db`);
projectDir = path.join(override, encoded);
} else {
dbPath = getAutoresearchDbPath(encoded);
projectDir = getAutoresearchProjectDir(encoded);
fs.mkdirSync(getAutoresearchDir(), { recursive: true });
}
const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd);
const cached = storageCache.get(dbPath);
if (cached) return cached;
fs.mkdirSync(path.dirname(dbPath), { recursive: true });
const storage = new AutoresearchStorage(dbPath, projectDir);
storageCache.set(dbPath, storage);
return storage;
}
export async function openAutoresearchStorageIfExists(cwd: string): Promise<AutoresearchStorage | null> {
const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd);
const cached = storageCache.get(dbPath);
if (cached) return cached;
if (!fs.existsSync(dbPath)) return null;
const storage = new AutoresearchStorage(dbPath, projectDir);
storageCache.set(dbPath, storage);
return storage;
}
async function resolveAutoresearchPaths(cwd: string): Promise<{ dbPath: string; projectDir: string }> {
const override = process.env.OMP_AUTORESEARCH_DB_DIR;
const repoRoot = (await git.repo.root(cwd)) ?? cwd;
const encoded = getEncodedProjectName(repoRoot);
if (override) {
return {
dbPath: path.join(override, `${encoded}.db`),
projectDir: path.join(override, encoded),
};
}
return {
dbPath: getAutoresearchDbPath(encoded),
projectDir: getAutoresearchProjectDir(encoded),
};
}
export function closeAllAutoresearchStorages(): void {
for (const storage of storageCache.values()) {
try {
@@ -1,3 +1,4 @@
import * as path from "node:path";
import { StringEnum } from "@oh-my-pi/pi-ai";
import { Text } from "@oh-my-pi/pi-tui";
import { Type } from "@sinclair/typebox";
@@ -5,11 +6,16 @@ import type { ToolDefinition } from "../../extensibility/extensions";
import type { Theme } from "../../modes/theme/theme";
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
import * as git from "../../utils/git";
import { parseWorkDirDirtyPaths } from "../git";
import { dedupeStrings, normalizePathSpec } from "../helpers";
import { buildExperimentState } from "../state";
import { openAutoresearchStorage, type SessionRow } from "../storage";
import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types";
export const HARNESS_FILENAME = "autoresearch.sh";
export const DEFAULT_HARNESS_COMMAND = `bash ${HARNESS_FILENAME}`;
const HARNESS_COMMIT_TITLE = "autoresearch: harness setup";
const initExperimentSchema = Type.Object({
name: Type.String({ description: "Human-readable experiment name." }),
goal: Type.Optional(Type.String({ description: "Free-form description of what this session optimizes." })),
@@ -23,12 +29,6 @@ const initExperimentSchema = Type.Object({
direction: Type.Optional(
StringEnum(["lower", "higher"], { description: "Whether lower or higher values are better. Defaults to lower." }),
),
preferred_command: Type.Optional(
Type.String({
description:
"Preferred benchmark command for this segment. Advisory; run_experiment accepts any command but warns when the command differs.",
}),
),
secondary_metrics: Type.Optional(
Type.Array(Type.String(), {
description: "Names of secondary metrics tracked alongside the primary metric.",
@@ -63,6 +63,8 @@ interface InitExperimentDetails {
createdSession: boolean;
bumpedSegment: boolean;
abandonedRuns: number;
harnessCommitted: boolean;
baselineCommit: string | null;
}
export function createInitExperimentTool(
@@ -72,7 +74,7 @@ export function createInitExperimentTool(
name: "init_experiment",
label: "Init Experiment",
description:
"Initialize or reconfigure the autoresearch session. Pass `new_segment: true` to start a fresh baseline within an existing session.",
"Initialize or reconfigure the autoresearch session. On first call (Phase 1 → Phase 2 transition), requires `./autoresearch.sh` to exist and pending harness changes are auto-committed on an autoresearch branch. Pass `new_segment: true` to start a fresh baseline within an existing session.",
parameters: initExperimentSchema,
defaultInactive: true,
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
@@ -85,29 +87,63 @@ export function createInitExperimentTool(
const offLimits = dedupeStrings((params.off_limits ?? []).map(normalizePathSpec));
const constraints = dedupeStrings(params.constraints ?? []);
const secondaryMetrics = dedupeStrings(params.secondary_metrics ?? []);
const preferredCommand = params.preferred_command?.trim() || null;
const goal = params.goal?.trim() || null;
const maxIterations =
params.max_iterations !== undefined && Number.isFinite(params.max_iterations) && params.max_iterations > 0
? Math.floor(params.max_iterations)
: null;
const branch = (await git.branch.current(ctx.cwd)) ?? null;
const onAutoresearchBranch = branch?.startsWith("autoresearch/") ?? false;
const existing = storage.getActiveSessionForBranch(branch);
const isNewSegmentInit = existing !== null && params.new_segment === true;
const requiresHarness = !existing || isNewSegmentInit;
if (requiresHarness) {
const harnessExists = await Bun.file(path.join(ctx.cwd, HARNESS_FILENAME)).exists();
if (!harnessExists) {
return {
content: [
{
type: "text",
text: `Error: ./${HARNESS_FILENAME} does not exist. Phase 1 of autoresearch is harness setup — write \`./${HARNESS_FILENAME}\` so it exits 0 and prints \`METRIC <name>=<value>\`, validate it via \`bash ${HARNESS_FILENAME}\`, then call init_experiment again.`,
},
],
};
}
}
let harnessCommitted = false;
let commitWarning: string | null = null;
if (requiresHarness && onAutoresearchBranch) {
const dirty = await detectPendingChanges(ctx.cwd);
if (dirty) {
try {
await git.stage.files(ctx.cwd, []);
const message = buildHarnessCommitMessage(goal, params.name);
await git.commit(ctx.cwd, message);
harnessCommitted = true;
} catch (err) {
commitWarning = `Failed to auto-commit harness changes: ${err instanceof Error ? err.message : String(err)}. Recording baseline at current HEAD; discard may not preserve uncommitted harness files.`;
}
}
}
const baselineCommit = await tryReadHeadSha(ctx.cwd);
const existing = storage.getActiveSession();
let session: SessionRow;
let createdSession = false;
let bumpedSegment = false;
let abandonedRuns = 0;
if (!existing) {
const baselineCommit = await tryReadHeadSha(ctx.cwd);
session = storage.openSession({
name: params.name,
goal,
primaryMetric: params.primary_metric,
metricUnit,
direction,
preferredCommand,
preferredCommand: DEFAULT_HARNESS_COMMAND,
branch,
baselineCommit,
maxIterations,
@@ -119,9 +155,8 @@ export function createInitExperimentTool(
createdSession = true;
} else {
abandonedRuns = storage.abandonPendingRuns(existing.id);
const updates = {
const updates: Parameters<typeof storage.updateSession>[1] = {
goal,
preferredCommand,
maxIterations,
scopePaths,
offLimits,
@@ -132,8 +167,11 @@ export function createInitExperimentTool(
direction,
branch,
};
if (isNewSegmentInit) {
updates.baselineCommit = baselineCommit;
}
let updated = storage.updateSession(existing.id, updates);
if (params.new_segment === true) {
if (isNewSegmentInit) {
updated = storage.bumpSegment(existing.id);
bumpedSegment = true;
}
@@ -159,6 +197,12 @@ export function createInitExperimentTool(
if (abandonedRuns > 0) {
lines.push(`Abandoned ${abandonedRuns} pending run${abandonedRuns === 1 ? "" : "s"} before reconfiguring.`);
}
if (harnessCommitted && session.baselineCommit) {
lines.push(`Committed harness setup at ${session.baselineCommit.slice(0, 12)}.`);
}
if (commitWarning) {
lines.push(commitWarning);
}
if (createdSession) {
lines.push(`Started session #${session.id}: ${session.name}`);
} else if (bumpedSegment) {
@@ -169,9 +213,7 @@ export function createInitExperimentTool(
lines.push(
`Metric: ${session.primaryMetric} (${session.metricUnit || "unitless"}, ${session.direction} is better)`,
);
if (session.preferredCommand) {
lines.push(`Preferred command: ${session.preferredCommand}`);
}
lines.push(`Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`);
if (session.scopePaths.length > 0) {
lines.push(`Files in scope: ${session.scopePaths.join(", ")}`);
}
@@ -188,10 +230,17 @@ export function createInitExperimentTool(
lines.push(`Baseline commit: ${session.baselineCommit.slice(0, 12)}`);
}
if (createdSession) {
lines.push("Run the baseline experiment now and log it.");
lines.push(
"Phase 2: iteration loop is active. Run the baseline experiment with `run_experiment` and log it.",
);
} else if (bumpedSegment) {
lines.push("Run a fresh baseline for the new segment.");
}
if (requiresHarness && !onAutoresearchBranch) {
lines.push(
"Note: not on a dedicated `autoresearch/*` branch — `log_experiment discard` will only revert run-modified files, not reset to baseline.",
);
}
return {
content: [{ type: "text", text: lines.join("\n") }],
@@ -200,6 +249,8 @@ export function createInitExperimentTool(
createdSession,
bumpedSegment,
abandonedRuns,
harnessCommitted,
baselineCommit: session.baselineCommit,
},
};
},
@@ -224,3 +275,23 @@ async function tryReadHeadSha(cwd: string): Promise<string | null> {
return null;
}
}
async function detectPendingChanges(cwd: string): Promise<boolean> {
try {
const statusText = await git.status(cwd, { porcelainV1: true, untrackedFiles: "all", z: true });
const workDirPrefix = await git.show.prefix(cwd).catch(() => "");
return parseWorkDirDirtyPaths(statusText, workDirPrefix).length > 0;
} catch {
return false;
}
}
function buildHarnessCommitMessage(goal: string | null, name: string): string {
const lines = [HARNESS_COMMIT_TITLE, "", `Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`];
if (goal) {
lines.push(`Goal: ${goal}`);
} else {
lines.push(`Session: ${name}`);
}
return lines.join("\n");
}
@@ -7,7 +7,7 @@ import type { ToolDefinition } from "../../extensibility/extensions";
import type { Theme } from "../../modes/theme/theme";
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
import * as git from "../../utils/git";
import { computeRunModifiedPaths, getCurrentAutoresearchBranch } from "../git";
import { computeRunModifiedPaths, getCurrentAutoresearchBranch, parseWorkDirDirtyPaths } from "../git";
import { ensureNumericMetricMap, formatNum, mergeAsi, pathMatchesSpec, sanitizeAsi } from "../helpers";
import {
buildExperimentState,
@@ -16,7 +16,7 @@ import {
findBaselineSecondary,
findBestKeptMetric,
} from "../state";
import { openAutoresearchStorage, type SessionRow } from "../storage";
import { openAutoresearchStorageIfExists, type SessionRow } from "../storage";
import type {
ASIData,
AutoresearchToolFactoryOptions,
@@ -81,14 +81,15 @@ export function createLogExperimentTool(
parameters: logExperimentSchema,
defaultInactive: true,
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
if (!session) {
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
if (!storage || !session) {
return {
content: [
{
type: "text",
text: "Error: no active autoresearch session. Call init_experiment first.",
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
},
],
};
@@ -113,8 +114,23 @@ export function createLogExperimentTool(
const branchName = await getCurrentAutoresearchBranch(options.pi, ctx.cwd);
const onAutoresearchBranch = branchName !== null;
const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths(ctx.cwd, pendingRun.preRunDirtyPaths);
const allModified = [...modifiedTracked, ...modifiedUntracked];
let allModified: string[];
if (onAutoresearchBranch) {
// On a dedicated autoresearch branch every iteration starts from a clean
// worktree (init_experiment baseline + previous keep commit / discard reset),
// so any currently-dirty path is the agent's iteration change. Off-branch we
// can't tell user dirt apart from agent edits, so we keep the (lossy)
// preRunDirtyPaths filter.
const statusText = await tryGitStatus(ctx.cwd);
const workDirPrefix = await tryGitPrefix(ctx.cwd);
allModified = parseWorkDirDirtyPaths(statusText, workDirPrefix);
} else {
const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths(
ctx.cwd,
pendingRun.preRunDirtyPaths,
);
allModified = [...modifiedTracked, ...modifiedUntracked];
}
const scopeDeviations = computeScopeDeviations(allModified, session);
const justification = params.justification?.trim() || null;
@@ -165,7 +181,6 @@ export function createLogExperimentTool(
ctx.cwd,
pendingRun.preRunDirtyPaths,
onAutoresearchBranch,
session.baselineCommit,
);
if (revertResult.error) {
return {
@@ -346,14 +361,15 @@ async function revertFailedExperiment(
cwd: string,
preRunDirtyPaths: string[],
onAutoresearchBranch: boolean,
baselineCommit: string | null,
): Promise<KeepCommitResult> {
if (onAutoresearchBranch) {
// Discard reverts only the current iteration's uncommitted changes — never
// rewinds prior `keep` commits. Reset to HEAD so any kept improvements
// already on the branch survive.
try {
const target = baselineCommit && baselineCommit.length > 0 ? baselineCommit : "HEAD";
await git.reset(cwd, { hard: true, target });
await git.reset(cwd, { hard: true, target: "HEAD" });
await git.clean(cwd);
return { note: `worktree reset to ${target.slice(0, 12)}` };
return { note: "worktree reset to HEAD" };
} catch (err) {
return { error: `git reset/clean failed: ${err instanceof Error ? err.message : String(err)}` };
}
@@ -7,7 +7,7 @@ import { Type } from "@sinclair/typebox";
import type { ToolDefinition } from "../../extensibility/extensions";
import type { Theme } from "../../modes/theme/theme";
import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output";
import { replaceTabs, shortenPath, truncateToWidth } from "../../tools/render-utils";
import { replaceTabs, shortenPath } from "../../tools/render-utils";
import * as git from "../../utils/git";
import { parseWorkDirDirtyPaths } from "../git";
import {
@@ -20,11 +20,11 @@ import {
parseMetricLines,
} from "../helpers";
import { buildExperimentState } from "../state";
import { openAutoresearchStorage } from "../storage";
import { openAutoresearchStorageIfExists } from "../storage";
import type { AutoresearchToolFactoryOptions, RunDetails, RunExperimentProgressDetails } from "../types";
import { DEFAULT_HARNESS_COMMAND } from "./init-experiment";
const runExperimentSchema = Type.Object({
command: Type.String({ description: "Shell command to run for this experiment." }),
timeout_seconds: Type.Optional(Type.Number({ description: "Timeout in seconds. Defaults to 600." })),
});
@@ -54,14 +54,15 @@ export function createRunExperimentTool(
parameters: runExperimentSchema,
defaultInactive: true,
async execute(_toolCallId, params, signal, onUpdate, ctx) {
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
if (!session) {
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
if (!storage || !session) {
return {
content: [
{
type: "text",
text: "Error: no active autoresearch session. Call init_experiment first.",
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
},
],
};
@@ -76,11 +77,7 @@ export function createRunExperimentTool(
return pending.id;
})();
let commandWarning: string | null = null;
if (session.preferredCommand && params.command.trim() !== session.preferredCommand.trim()) {
commandWarning = `Note: command differs from preferred (\`${session.preferredCommand}\`). Re-init the experiment if the workload itself changed.`;
}
const resolvedCommand = DEFAULT_HARNESS_COMMAND;
const preRunStatus = await tryGitStatus(ctx.cwd);
const workDirPrefix = await tryGitPrefix(ctx.cwd);
const preRunDirtyPaths = parseWorkDirDirtyPaths(preRunStatus, workDirPrefix);
@@ -89,7 +86,7 @@ export function createRunExperimentTool(
const insertedRun = storage.insertRun({
sessionId: session.id,
segment: session.currentSegment,
command: params.command,
command: resolvedCommand,
logPath: "", // patched after we know the run id
preRunDirtyPaths,
startedAt,
@@ -107,7 +104,7 @@ export function createRunExperimentTool(
runtime.lastRunSummary = null;
runtime.runningExperiment = {
startedAt,
command: params.command,
command: resolvedCommand,
runDirectory,
runNumber: insertedRun.id,
};
@@ -118,7 +115,7 @@ export function createRunExperimentTool(
let execution: ProcessExecutionResult;
try {
execution = await executeProcess({
command: ["bash", "-lc", params.command],
command: ["bash", "-lc", resolvedCommand],
cwd: ctx.cwd,
logPath: benchmarkLogPath,
timeoutMs,
@@ -178,7 +175,7 @@ export function createRunExperimentTool(
runNumber: insertedRun.id,
runDirectory,
benchmarkLogPath,
command: params.command,
command: resolvedCommand,
exitCode: execution.exitCode,
durationSeconds,
passed,
@@ -191,14 +188,13 @@ export function createRunExperimentTool(
metricName: session.primaryMetric,
metricUnit: session.metricUnit,
preRunDirtyPaths,
commandWarning,
abandonedPriorRun,
truncation: llmTruncation.truncated ? llmTruncation : undefined,
fullOutputPath: execution.logPath,
};
runtime.lastRunSummary = {
command: params.command,
command: resolvedCommand,
durationSeconds,
parsedAsi,
parsedMetrics,
@@ -222,7 +218,6 @@ export function createRunExperimentTool(
options.dashboard.requestRender();
const headerLines: string[] = [];
if (commandWarning) headerLines.push(commandWarning);
if (abandonedPriorRun !== null) {
headerLines.push(`Note: abandoned prior pending run #${abandonedPriorRun} before starting this run.`);
}
@@ -238,10 +233,9 @@ export function createRunExperimentTool(
details: resultDetails,
};
},
renderCall(args, _options, theme): Text {
const commandPreview = truncateToWidth(replaceTabs(args.command), 100);
renderCall(_args, _options, theme): Text {
return new Text(
`${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", commandPreview)}`,
`${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", DEFAULT_HARNESS_COMMAND)}`,
0,
0,
);
@@ -3,8 +3,9 @@ import { Type } from "@sinclair/typebox";
import type { ToolDefinition } from "../../extensibility/extensions";
import type { Theme } from "../../modes/theme/theme";
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
import * as git from "../../utils/git";
import { buildExperimentState } from "../state";
import { openAutoresearchStorage } from "../storage";
import { openAutoresearchStorageIfExists } from "../storage";
import type { AutoresearchToolFactoryOptions } from "../types";
const updateNotesSchema = Type.Object({
@@ -34,14 +35,15 @@ export function createUpdateNotesTool(
parameters: updateNotesSchema,
defaultInactive: true,
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
const storage = await openAutoresearchStorage(ctx.cwd);
const session = storage.getActiveSession();
if (!session) {
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
if (!storage || !session) {
return {
content: [
{
type: "text",
text: "Error: no active autoresearch session. Call init_experiment first.",
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
},
],
};
@@ -51,7 +51,6 @@ export interface ExperimentState {
currentSegment: number;
maxExperiments: number | null;
confidence: number | null;
benchmarkCommand: string | null;
scopePaths: string[];
offLimits: string[];
constraints: string[];
@@ -86,7 +85,6 @@ export interface RunDetails {
metricName: string;
metricUnit: string;
preRunDirtyPaths: string[];
commandWarning: string | null;
abandonedPriorRun: number | null;
truncation?: TruncationResult;
fullOutputPath?: string;
@@ -84,6 +84,10 @@ async function checkoutBranch(dir: string, name: string): Promise<void> {
await $`git checkout -b ${name}`.cwd(dir).quiet();
}
async function writeHarnessStub(dir: string, body = "echo METRIC m=1"): Promise<void> {
await Bun.write(path.join(dir, "autoresearch.sh"), `#!/usr/bin/env bash\n${body}\n`);
}
describe("init_experiment", () => {
let dbOverride: string;
@@ -99,6 +103,7 @@ describe("init_experiment", () => {
it("opens a new session and persists scope and metric metadata", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -114,7 +119,6 @@ describe("init_experiment", () => {
primary_metric: "runtime_ms",
metric_unit: "ms",
direction: "lower",
preferred_command: "bun bench",
scope_paths: ["src", "src/foo"],
off_limits: ["test"],
secondary_metrics: ["memory_mb"],
@@ -141,6 +145,7 @@ describe("init_experiment", () => {
it("updates fields without bumping segment when no new_segment flag is passed", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -171,6 +176,7 @@ describe("init_experiment", () => {
it("bumps segment when new_segment is true on a re-init", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -188,6 +194,78 @@ describe("init_experiment", () => {
expect(result.details?.bumpedSegment).toBe(true);
expect(result.details?.state.currentSegment).toBe(1);
});
it("rejects when autoresearch.sh is missing on first init", async () => {
const dir = makeTempDir();
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await tool.execute(
"call-1",
{ name: "x", primary_metric: "m" },
undefined,
undefined,
createCtx(dir),
);
expect(firstTextBlockText(result.content)).toContain("autoresearch.sh");
const storage = await openAutoresearchStorage(dir);
expect(storage.getActiveSession()).toBeNull();
});
it("auto-commits pending harness changes on an autoresearch branch", async () => {
const dir = makeTempDir();
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
await checkoutBranch(dir, "autoresearch/setup-test");
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await tool.execute(
"call-1",
{ name: "x", primary_metric: "m", goal: "speed" },
undefined,
undefined,
createCtx(dir),
);
expect(result.details?.harnessCommitted).toBe(true);
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
expect(newHead).not.toBe(initialBaseline);
expect(result.details?.baselineCommit).toBe(newHead);
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
expect(status).toBe("");
const message = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim();
expect(message).toContain("autoresearch: harness setup");
});
it("does not auto-commit when not on an autoresearch branch", async () => {
const dir = makeTempDir();
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await tool.execute(
"call-1",
{ name: "x", primary_metric: "m" },
undefined,
undefined,
createCtx(dir),
);
expect(result.details?.harnessCommitted).toBe(false);
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
expect(newHead).toBe(initialBaseline);
// Harness file is still in the worktree, untracked.
expect(fs.existsSync(path.join(dir, "autoresearch.sh"))).toBe(true);
});
});
describe("run_experiment", () => {
@@ -211,12 +289,13 @@ describe("run_experiment", () => {
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await run.execute("call-1", { command: "echo hi" }, undefined, undefined, createCtx(dir));
const result = await run.execute("call-1", {}, undefined, undefined, createCtx(dir));
expect(firstTextBlockText(result.content)).toContain("no active autoresearch session");
});
it("accepts arbitrary commands, parses METRIC/ASI, and stores a run", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir, "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline");
const runtime = createSessionRuntime();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -235,16 +314,7 @@ describe("run_experiment", () => {
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await run.execute(
"r",
{
command: "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline",
timeout_seconds: 5,
},
undefined,
undefined,
createCtx(dir),
);
const result = await run.execute("r", { timeout_seconds: 5 }, undefined, undefined, createCtx(dir));
const details = result.details as RunDetails;
expect(details.parsedPrimary).toBe(42);
expect(details.parsedMetrics).toMatchObject({ runtime_ms: 42, memory_mb: 12 });
@@ -262,6 +332,7 @@ describe("run_experiment", () => {
it("abandons a prior pending run instead of blocking", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const initTool = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -274,37 +345,32 @@ describe("run_experiment", () => {
getRuntime: () => runtime,
pi: createPiHarness().api,
});
await run.execute("r1", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir));
const result = await run.execute("r2", { command: "echo METRIC m=2" }, undefined, undefined, createCtx(dir));
await run.execute("r1", {}, undefined, undefined, createCtx(dir));
const result = await run.execute("r2", {}, undefined, undefined, createCtx(dir));
const details = result.details as RunDetails;
expect(details.abandonedPriorRun).not.toBeNull();
expect(details.runNumber).not.toBe(details.abandonedPriorRun);
});
it("warns when command differs from the preferred command", async () => {
it("runs ./autoresearch.sh and parses METRIC/ASI from its output", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir, "echo METRIC m=99");
const runtime = createSessionRuntime();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
await init.execute(
"i",
{ name: "x", primary_metric: "m", preferred_command: "echo preferred" },
undefined,
undefined,
createCtx(dir),
);
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await run.execute("r", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir));
const result = await run.execute("r", {}, undefined, undefined, createCtx(dir));
const details = result.details as RunDetails;
expect(details.commandWarning).toContain("preferred");
expect(firstTextBlockText(result.content)).toContain("preferred");
expect(details.command).toBe("bash autoresearch.sh");
expect(details.parsedPrimary).toBe(99);
});
});
@@ -322,6 +388,7 @@ describe("log_experiment", () => {
});
async function setupRun(dir: string, runtime = createSessionRuntime()) {
await writeHarnessStub(dir, "echo METRIC runtime_ms=10");
const harness = createPiHarness();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -346,7 +413,7 @@ describe("log_experiment", () => {
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", { command: "echo METRIC runtime_ms=10" }, undefined, undefined, createCtx(dir));
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const log = createLogExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
@@ -357,6 +424,7 @@ describe("log_experiment", () => {
it("rejects when no pending run exists", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
@@ -480,7 +548,7 @@ describe("log_experiment", () => {
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r2", { command: "echo METRIC runtime_ms=8" }, undefined, undefined, createCtx(dir));
await run.execute("r2", {}, undefined, undefined, createCtx(dir));
const log2 = createLogExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
@@ -512,6 +580,7 @@ describe("log_experiment", () => {
it("on a non-autoresearch branch, discard reverts only run-modified files", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
await initGitRepo(dir);
// Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
@@ -539,7 +608,7 @@ describe("log_experiment", () => {
});
// Pre-existing untracked file (will not be touched by revert because it was dirty before run)
await Bun.write(path.join(dir, "preexisting.txt"), "leave me\n");
await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir));
await run.execute("r", {}, undefined, undefined, createCtx(dir));
// Simulate a run-introduced change
await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 2;\n");
await Bun.write(path.join(dir, "src", "new.ts"), "export const NEW = true;\n");
@@ -564,9 +633,13 @@ describe("log_experiment", () => {
expect(fs.readFileSync(path.join(dir, "src", "edit-me.ts"), "utf8")).toBe("export const v = 1;\n");
});
it("on an autoresearch branch, discard resets the worktree to baseline_commit", async () => {
it("on an autoresearch branch, discard reverts uncommitted changes but preserves prior commits", async () => {
const dir = makeTempDir();
const { baselineCommit } = await initGitRepo(dir);
await initGitRepo(dir);
// Commit the harness on main so it is part of the autoresearch branch's baseline.
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/test-20260501");
const runtime = createSessionRuntime();
const harness = createPiHarness();
@@ -576,16 +649,21 @@ describe("log_experiment", () => {
pi: harness.api,
});
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
// Simulate a previously kept iteration by committing it directly on the branch.
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m "kept iteration"`.cwd(dir).quiet();
const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir));
// Modify and commit a file on the autoresearch branch — discard should reset HEAD back to baseline.
await Bun.write(path.join(dir, "src", "stub.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m wip`.cwd(dir).quiet();
await run.execute("r", {}, undefined, undefined, createCtx(dir));
// Current iteration's uncommitted edits.
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 999;\n");
await Bun.write(path.join(dir, "scratch.ts"), "// junk\n");
const log = createLogExperimentTool({
dashboard: dashboardStub(),
@@ -599,9 +677,117 @@ describe("log_experiment", () => {
undefined,
createCtx(dir),
);
const headSha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
expect(headSha).toBe(baselineCommit);
expect(fs.existsSync(path.join(dir, "src", "stub.ts"))).toBe(false);
const headAfter = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
// Prior commits survive — discard does not rewind history.
expect(headAfter).toBe(headBeforeDiscard);
// Uncommitted iteration changes are gone.
expect(fs.readFileSync(path.join(dir, "src", "kept.ts"), "utf8")).toBe("export const v = 1;\n");
expect(fs.existsSync(path.join(dir, "scratch.ts"))).toBe(false);
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
expect(status).toBe("");
});
it("on an autoresearch branch, keep commits files that were dirty before run_experiment", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
// Seed a tracked file that the agent will edit during the iteration.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m seed`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/keep-test");
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await init.execute(
"i",
{ name: "x", primary_metric: "m", scope_paths: ["src"] },
undefined,
undefined,
createCtx(dir),
);
// Agent edits BEFORE running the benchmark — the iteration's diff is dirty
// at run_experiment time.
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 2;\n");
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const log = createLogExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
const result = await log.execute(
"l",
{ metric: 42, status: "keep", description: "improvement" },
undefined,
undefined,
createCtx(dir),
);
const details = result.details as LogDetails;
expect(details.experiment.modifiedPaths).toContain("src/store.ts");
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
expect(status).toBe("");
const lastMsg = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim();
expect(lastMsg).toContain("improvement");
});
it("flags off-scope dirty files even when they were dirty before run_experiment", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/scope-test");
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await init.execute(
"i",
{ name: "x", primary_metric: "m", scope_paths: ["src"], off_limits: ["forbidden"] },
undefined,
undefined,
createCtx(dir),
);
// Off-scope edit BEFORE run_experiment.
fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true });
await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n");
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const log = createLogExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
const result = await log.execute(
"l",
{ metric: 42, status: "keep", description: "off-scope" },
undefined,
undefined,
createCtx(dir),
);
const details = result.details as LogDetails;
expect(details.scopeDeviations).toContain("forbidden/x.ts");
});
});
@@ -620,6 +806,7 @@ describe("update_notes", () => {
it("replaces session notes and refreshes runtime state", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({