feat(coding-agent/autoresearch): added branch-specific state restore
- Added branch-aware session loading so autoresearch state only rehydrates for current branch. - Replaced user-specified experiment commands with fixed `bash autoresearch.sh` execution flow. - Enforced safer setup checks, including missing `autoresearch.sh` and uncommitted-worktree errors. - Added branch-specific storage helpers, baseline-commit persistence, and expanded tests for dirty-path cases.
This commit is contained in:
@@ -67,10 +67,8 @@ export async function ensureAutoresearchBranch(
|
||||
if (dirtyPaths.length > 0) {
|
||||
const preview = formatDirtyPaths(dirtyPaths);
|
||||
return {
|
||||
ok: true,
|
||||
branchName: null,
|
||||
created: false,
|
||||
warning: `Worktree is dirty (${preview}). Continuing on the current branch; auto-commit and full-tree reset are disabled until you commit/stash these changes.`,
|
||||
ok: false,
|
||||
error: `Worktree is dirty (${preview}). Commit or stash these changes before starting autoresearch — a fresh autoresearch/* branch needs a clean baseline.`,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ import { createDashboardController } from "./dashboard";
|
||||
import { ensureAutoresearchBranch } from "./git";
|
||||
import { formatNum } from "./helpers";
|
||||
import promptTemplate from "./prompt.md" with { type: "text" };
|
||||
import setupPromptTemplate from "./prompt-setup.md" with { type: "text" };
|
||||
import resumeMessageTemplate from "./resume-message.md" with { type: "text" };
|
||||
import {
|
||||
buildExperimentState,
|
||||
@@ -20,7 +21,7 @@ import {
|
||||
findBestKeptMetric,
|
||||
reconstructControlState,
|
||||
} from "./state";
|
||||
import { openAutoresearchStorage, type RunRow } from "./storage";
|
||||
import { openAutoresearchStorage, openAutoresearchStorageIfExists, type RunRow, type SessionRow } from "./storage";
|
||||
import { createInitExperimentTool } from "./tools/init-experiment";
|
||||
import { createLogExperimentTool } from "./tools/log-experiment";
|
||||
import { createRunExperimentTool } from "./tools/run-experiment";
|
||||
@@ -36,21 +37,49 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
const getSessionKey = (ctx: ExtensionContext): string => ctx.sessionManager.getSessionId();
|
||||
const getRuntime = (ctx: ExtensionContext): AutoresearchRuntime => runtimeStore.ensure(getSessionKey(ctx));
|
||||
|
||||
const loadActiveSession = async (
|
||||
ctx: ExtensionContext,
|
||||
): Promise<{ session: SessionRow | null; currentBranch: string | null }> => {
|
||||
const currentBranch = await tryReadBranch(ctx.cwd);
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
if (!storage) return { session: null, currentBranch };
|
||||
const session = storage.getActiveSessionForBranch(currentBranch);
|
||||
return { session, currentBranch };
|
||||
};
|
||||
|
||||
const rehydrate = async (ctx: ExtensionContext): Promise<void> => {
|
||||
const runtime = getRuntime(ctx);
|
||||
const control = reconstructControlState(ctx.sessionManager.getBranch());
|
||||
runtime.goal = control.goal;
|
||||
runtime.autoresearchMode = control.autoresearchMode;
|
||||
runtime.autoResumeArmed = false;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
if (session) {
|
||||
const loggedRuns = storage.listLoggedRuns(session.id);
|
||||
runtime.state = buildExperimentState(session, loggedRuns);
|
||||
runtime.goal = runtime.goal ?? session.goal;
|
||||
runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id));
|
||||
// Skip storage entirely if autoresearch was never activated in this conversation.
|
||||
// This is the common case: every project gets a session_start event but most
|
||||
// never touch autoresearch, so we must not create a SQLite file just to look.
|
||||
const everActivated = control.lastMode !== null;
|
||||
const { session, currentBranch } = everActivated
|
||||
? await loadActiveSession(ctx)
|
||||
: { session: null, currentBranch: null };
|
||||
|
||||
// Mode is effective only when the recorded session matches the current git
|
||||
// branch. When the user switches off the autoresearch branch the widget hides
|
||||
// and the experiment tools detach, but the session entries are preserved so
|
||||
// switching back resumes seamlessly.
|
||||
const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch;
|
||||
runtime.autoresearchMode = control.autoresearchMode && onActiveBranch;
|
||||
|
||||
if (session && onActiveBranch) {
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
if (storage) {
|
||||
const loggedRuns = storage.listLoggedRuns(session.id);
|
||||
runtime.state = buildExperimentState(session, loggedRuns);
|
||||
runtime.goal = runtime.goal ?? session.goal;
|
||||
runtime.lastRunSummary = pendingRunSummaryFromRow(storage.getPendingRun(session.id));
|
||||
} else {
|
||||
runtime.state = createExperimentState();
|
||||
runtime.lastRunSummary = null;
|
||||
}
|
||||
} else {
|
||||
runtime.state = createExperimentState();
|
||||
runtime.lastRunSummary = null;
|
||||
@@ -151,8 +180,12 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
ctx.ui.notify(branchResult.warning, "warning");
|
||||
}
|
||||
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const existingSession = storage.getActiveSession();
|
||||
// Look up an existing session for the branch we just landed on. A session
|
||||
// recorded under a different autoresearch/* branch is intentionally ignored
|
||||
// — `/autoresearch` on a fresh branch starts a fresh session. Only open the
|
||||
// DB if it already exists; the empty-state path must not create one.
|
||||
const existingStorage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
const existingSession = existingStorage?.getActiveSessionForBranch(branchResult.branchName) ?? null;
|
||||
const resumeContext = trimmed;
|
||||
const branchStatusLine = branchResult.branchName
|
||||
? branchResult.created
|
||||
@@ -160,13 +193,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
: `Using dedicated git branch \`${branchResult.branchName}\`.`
|
||||
: "Continuing on the current branch — no autoresearch branch was created.";
|
||||
|
||||
if (existingSession) {
|
||||
if (goalArg) storage.updateSession(existingSession.id, { goal: goalArg });
|
||||
if (existingSession && existingStorage) {
|
||||
if (goalArg) existingStorage.updateSession(existingSession.id, { goal: goalArg });
|
||||
if (branchResult.branchName) {
|
||||
storage.updateSession(existingSession.id, { branch: branchResult.branchName });
|
||||
existingStorage.updateSession(existingSession.id, { branch: branchResult.branchName });
|
||||
}
|
||||
const refreshed = storage.getSessionById(existingSession.id) ?? existingSession;
|
||||
runtime.state = buildExperimentState(refreshed, storage.listLoggedRuns(refreshed.id));
|
||||
const refreshed = existingStorage.getSessionById(existingSession.id) ?? existingSession;
|
||||
runtime.state = buildExperimentState(refreshed, existingStorage.listLoggedRuns(refreshed.id));
|
||||
runtime.goal = refreshed.goal ?? goalArg;
|
||||
setMode(ctx, true, runtime.goal, "on");
|
||||
dashboard.updateWidget(ctx, runtime);
|
||||
@@ -231,9 +264,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
runtime.autoResumeArmed = false;
|
||||
return;
|
||||
}
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
const pendingRow = session ? storage.getPendingRun(session.id) : null;
|
||||
const { session } = await loadActiveSession(ctx);
|
||||
const storage = session ? await openAutoresearchStorageIfExists(ctx.cwd) : null;
|
||||
const pendingRow = session && storage ? storage.getPendingRun(session.id) : null;
|
||||
const pendingRun = pendingRunSummaryFromRow(pendingRow);
|
||||
runtime.lastRunSummary = pendingRun;
|
||||
runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration;
|
||||
@@ -261,12 +294,27 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
api.on("before_agent_start", async (event, ctx) => {
|
||||
const runtime = getRuntime(ctx);
|
||||
if (!runtime.autoresearchMode) return;
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
if (session) {
|
||||
// Re-check git branch on every agent start. If the user manually switched
|
||||
// off the autoresearch/* branch between turns, we silently drop autoresearch
|
||||
// from this turn — the widget hides, the experiment tools detach, and we do
|
||||
// not inject the autoresearch system prompt.
|
||||
const { session, currentBranch } = await loadActiveSession(ctx);
|
||||
const onActiveBranch = session === null || session.branch === null || session.branch === currentBranch;
|
||||
if (!onActiveBranch) {
|
||||
runtime.autoresearchMode = false;
|
||||
runtime.state = createExperimentState();
|
||||
runtime.lastRunSummary = null;
|
||||
runtime.runningExperiment = null;
|
||||
dashboard.updateWidget(ctx, runtime);
|
||||
const experimentTools = new Set(EXPERIMENT_TOOL_NAMES);
|
||||
await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name)));
|
||||
return;
|
||||
}
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
if (session && storage) {
|
||||
runtime.state = buildExperimentState(session, storage.listLoggedRuns(session.id));
|
||||
}
|
||||
const pendingRow = session ? storage.getPendingRun(session.id) : null;
|
||||
const pendingRow = session && storage ? storage.getPendingRun(session.id) : null;
|
||||
const pendingRun = pendingRunSummaryFromRow(pendingRow);
|
||||
runtime.lastRunSummary = pendingRun;
|
||||
runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration;
|
||||
@@ -301,9 +349,25 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
run_number: r.runNumber,
|
||||
paths: r.scopeDeviations.join(", "),
|
||||
}));
|
||||
const lastCommand = pendingRun?.command ?? null;
|
||||
const showCommandWarning =
|
||||
Boolean(state.benchmarkCommand) && lastCommand !== null && lastCommand !== state.benchmarkCommand;
|
||||
if (!session) {
|
||||
const currentBranch = await tryReadBranch(ctx.cwd);
|
||||
const onAutoresearchBranch = currentBranch?.startsWith("autoresearch/") ?? false;
|
||||
const baselineWarning = onAutoresearchBranch
|
||||
? null
|
||||
: "Heads up: you are not on a dedicated `autoresearch/*` branch. `log_experiment discard` will only revert run-modified files, not reset to baseline — so harness files written before `init_experiment` may not survive a discard. Clean the worktree and re-run `/autoresearch` if you want full revert safety.";
|
||||
return {
|
||||
systemPrompt: prompt.render(setupPromptTemplate, {
|
||||
base_system_prompt: event.systemPrompt,
|
||||
has_goal: goal.trim().length > 0,
|
||||
goal,
|
||||
working_dir: ctx.cwd,
|
||||
has_branch: Boolean(currentBranch),
|
||||
branch: currentBranch ?? "",
|
||||
has_baseline_warning: baselineWarning !== null,
|
||||
baseline_warning: baselineWarning ?? "",
|
||||
}),
|
||||
};
|
||||
}
|
||||
return {
|
||||
systemPrompt: prompt.render(promptTemplate, {
|
||||
base_system_prompt: event.systemPrompt,
|
||||
@@ -339,9 +403,6 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined
|
||||
? formatNum(pendingRun.parsedPrimary, state.metricUnit)
|
||||
: null,
|
||||
has_preferred_command_warning: showCommandWarning,
|
||||
preferred_command: state.benchmarkCommand ?? "",
|
||||
last_command: lastCommand ?? "",
|
||||
}),
|
||||
};
|
||||
});
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
{{base_system_prompt}}
|
||||
|
||||
## Autoresearch Mode — Phase 1: Harness Setup
|
||||
|
||||
Autoresearch mode is active and there is no session yet. Your job in this turn is to **build the benchmark harness**, not to optimise anything. Optimisation starts only after you call `init_experiment`.
|
||||
|
||||
{{#if has_goal}}
|
||||
Primary goal (for context — implement the harness so it can measure this):
|
||||
{{goal}}
|
||||
{{else}}
|
||||
There is no goal recorded yet. Infer what to optimise from the latest user message and design the harness to measure that. Capture the goal when you call `init_experiment`.
|
||||
{{/if}}
|
||||
|
||||
Working directory: `{{working_dir}}`
|
||||
{{#if has_branch}}Active branch: `{{branch}}`{{/if}}
|
||||
{{#if has_baseline_warning}}
|
||||
|
||||
{{baseline_warning}}
|
||||
{{/if}}
|
||||
|
||||
### What you must produce
|
||||
|
||||
Write `./autoresearch.sh` at the working directory. It is the canonical benchmark entrypoint and must:
|
||||
|
||||
- exit 0 on success and non-zero on failure;
|
||||
- print the primary metric as a single line `METRIC <name>=<value>`;
|
||||
- print any secondary metrics as additional `METRIC <name>=<value>` lines;
|
||||
- run the same workload deterministically every time (no live network, no time-of-day dependencies, fixed seeds where applicable).
|
||||
|
||||
You **may** edit anything else needed to make `autoresearch.sh` work — benchmark binaries, `Cargo.toml`, `package.json`, helper scripts, fixtures. All those edits are part of the harness baseline and will be committed for you when you call `init_experiment` on an autoresearch branch.
|
||||
|
||||
### Steps
|
||||
|
||||
1. Inspect the target. Read source, identify what to measure, decide on the workload.
|
||||
2. Write `autoresearch.sh` plus any supporting files (benchmark binaries, fixtures, etc.).
|
||||
3. Validate it: invoke `bash autoresearch.sh` through the regular `bash` tool. Confirm it exits 0 and emits at least one `METRIC` line. Iterate on the harness until it does.
|
||||
4. Call `init_experiment` with the goal, primary metric (matching the `METRIC` name), and scope. This snapshots the worktree as the baseline and starts Phase 2 (the iteration loop).
|
||||
|
||||
### Rules
|
||||
|
||||
- Do **not** call `run_experiment`, `log_experiment`, or `update_notes` yet. They will error with "no active autoresearch session" until `init_experiment` runs.
|
||||
- Do **not** treat a compile-only check as a benchmark. The harness must actually execute the workload and emit `METRIC`.
|
||||
- Do **not** create `autoresearch.md`, `autoresearch.checks.sh`, `autoresearch.program.md`, `autoresearch.ideas.md`, `autoresearch.jsonl`, `.autoresearch/`, or `autoresearch.config.json`. Session state is tracked for you.
|
||||
@@ -11,7 +11,7 @@ Primary goal:
|
||||
There is no goal recorded for this session yet. Infer what to optimize from the latest user message and the conversation; capture the goal in your notes (`update_notes`) once it is clear.
|
||||
{{/if}}
|
||||
|
||||
Session state and run artifacts are managed for you. Do not create `autoresearch.md`, `autoresearch.sh`, or `.autoresearch/` in this repo.
|
||||
Session state and run artifacts are managed for you. The benchmark entrypoint is `bash autoresearch.sh` (committed during Phase 1). Do not edit `autoresearch.sh` mid-segment unless you intentionally bump segment via `init_experiment new_segment: true`. Do not create `autoresearch.md` or `.autoresearch/` in this repo.
|
||||
|
||||
Working directory: `{{working_dir}}`
|
||||
{{#if has_branch}}Active branch: `{{branch}}`{{/if}}
|
||||
@@ -21,13 +21,13 @@ You are running an autonomous experiment loop. Keep iterating until the user int
|
||||
|
||||
### Available tools
|
||||
- `init_experiment` — open or reconfigure the session. Pass `new_segment: true` to start a fresh baseline within the current session.
|
||||
- `run_experiment` — run any benchmark command. Pass the actual command; output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the command are parsed back to you.
|
||||
- `run_experiment` — run the benchmark (`bash autoresearch.sh`). Output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the harness are parsed back to you. The command is fixed; if you need a different workload, edit `autoresearch.sh` and bump segment via `init_experiment new_segment: true`.
|
||||
- `log_experiment` — record the result. On `keep`, modified files are committed for you; on `discard`/`crash`/`checks_failed`, the worktree is reverted. Pass `flag_runs` to mark earlier runs as suspect; flagged runs are excluded from baseline and best-metric math.
|
||||
- `update_notes` — replace the durable session playbook (`body`) or append to the ideas backlog (`append_idea`). The notes are injected into your system prompt every iteration.
|
||||
|
||||
### Operating protocol
|
||||
1. Understand the target before touching code: read source, identify the bottleneck, verify prerequisites and benchmark inputs.
|
||||
2. Capture goal, benchmark command, primary metric, scope, and constraints in `init_experiment`. Update them later via another `init_experiment` call (no segment bump) or via `update_notes`.
|
||||
2. Update goal, scope, or constraints via another `init_experiment` call (no segment bump) or `update_notes`. Bump segment when you intentionally change `autoresearch.sh`.
|
||||
3. Establish a baseline first.
|
||||
4. Iterate: change code, run `run_experiment`, log honestly with `log_experiment`. One coherent experiment per iteration.
|
||||
5. Keep the primary metric as the decision maker:
|
||||
@@ -95,11 +95,6 @@ An unlogged run is waiting:
|
||||
|
||||
Finish the `log_experiment` step before starting another benchmark.
|
||||
{{/if}}
|
||||
{{#if has_preferred_command_warning}}
|
||||
|
||||
### Preferred command
|
||||
Last `run_experiment` used `{{last_command}}`. Preferred command for this segment is `{{preferred_command}}`. If the workload changed intentionally, call `init_experiment` to update the preferred command (or pass `new_segment: true` to start a fresh baseline).
|
||||
{{/if}}
|
||||
|
||||
### Guardrails
|
||||
- Do not game the benchmark.
|
||||
|
||||
@@ -26,7 +26,6 @@ export function createExperimentState(): ExperimentState {
|
||||
currentSegment: 0,
|
||||
maxExperiments: null,
|
||||
confidence: null,
|
||||
benchmarkCommand: null,
|
||||
scopePaths: [],
|
||||
offLimits: [],
|
||||
constraints: [],
|
||||
@@ -177,7 +176,6 @@ export function buildExperimentState(session: SessionRow, loggedRuns: RunRow[]):
|
||||
state.metricName = session.primaryMetric;
|
||||
state.metricUnit = session.metricUnit;
|
||||
state.bestDirection = session.direction;
|
||||
state.benchmarkCommand = session.preferredCommand;
|
||||
state.scopePaths = [...session.scopePaths];
|
||||
state.offLimits = [...session.offLimits];
|
||||
state.constraints = [...session.constraints];
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { Database, type SQLQueryBindings } from "bun:sqlite";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { getAutoresearchDbPath, getAutoresearchDir, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils";
|
||||
import { getAutoresearchDbPath, getAutoresearchProjectDir, logger } from "@oh-my-pi/pi-utils";
|
||||
import { getEncodedProjectName } from "../task/worktree";
|
||||
import * as git from "../utils/git";
|
||||
import type { ASIData, ExperimentStatus, MetricDirection, NumericMetricMap } from "./types";
|
||||
@@ -86,6 +86,7 @@ export interface UpdateSessionParams {
|
||||
metricUnit?: string;
|
||||
direction?: MetricDirection;
|
||||
branch?: string | null;
|
||||
baselineCommit?: string | null;
|
||||
notes?: string;
|
||||
}
|
||||
|
||||
@@ -278,6 +279,24 @@ export class AutoresearchStorage {
|
||||
return row ? rowToSession(row) : null;
|
||||
}
|
||||
|
||||
getActiveSessionForBranch(branch: string | null): SessionRow | null {
|
||||
// Most-recent active session whose recorded branch matches the caller's branch.
|
||||
// `branch === null` means "no git repo / no branch info" — treat null on both
|
||||
// sides as a match.
|
||||
if (branch === null) {
|
||||
const stmt = this.#db.prepare<SessionDbRow, []>(
|
||||
"SELECT * FROM sessions WHERE closed_at IS NULL AND branch IS NULL ORDER BY id DESC LIMIT 1",
|
||||
);
|
||||
const row = stmt.get();
|
||||
return row ? rowToSession(row) : null;
|
||||
}
|
||||
const stmt = this.#db.prepare<SessionDbRow, [string]>(
|
||||
"SELECT * FROM sessions WHERE closed_at IS NULL AND branch = ? ORDER BY id DESC LIMIT 1",
|
||||
);
|
||||
const row = stmt.get(branch);
|
||||
return row ? rowToSession(row) : null;
|
||||
}
|
||||
|
||||
getSessionById(sessionId: number): SessionRow | null {
|
||||
const stmt = this.#db.prepare<SessionDbRow, [number]>("SELECT * FROM sessions WHERE id = ?");
|
||||
const row = stmt.get(sessionId);
|
||||
@@ -362,6 +381,10 @@ export class AutoresearchStorage {
|
||||
setClauses.push("branch = ?");
|
||||
values.push(updates.branch);
|
||||
}
|
||||
if (updates.baselineCommit !== undefined) {
|
||||
setClauses.push("baseline_commit = ?");
|
||||
values.push(updates.baselineCommit);
|
||||
}
|
||||
if (updates.notes !== undefined) {
|
||||
setClauses.push("notes = ?");
|
||||
values.push(updates.notes);
|
||||
@@ -516,27 +539,41 @@ export class AutoresearchStorage {
|
||||
const storageCache = new Map<string, AutoresearchStorage>();
|
||||
|
||||
export async function openAutoresearchStorage(cwd: string): Promise<AutoresearchStorage> {
|
||||
const override = process.env.OMP_AUTORESEARCH_DB_DIR;
|
||||
const repoRoot = (await git.repo.root(cwd)) ?? cwd;
|
||||
const encoded = getEncodedProjectName(repoRoot);
|
||||
let dbPath: string;
|
||||
let projectDir: string;
|
||||
if (override) {
|
||||
fs.mkdirSync(override, { recursive: true });
|
||||
dbPath = path.join(override, `${encoded}.db`);
|
||||
projectDir = path.join(override, encoded);
|
||||
} else {
|
||||
dbPath = getAutoresearchDbPath(encoded);
|
||||
projectDir = getAutoresearchProjectDir(encoded);
|
||||
fs.mkdirSync(getAutoresearchDir(), { recursive: true });
|
||||
}
|
||||
const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd);
|
||||
const cached = storageCache.get(dbPath);
|
||||
if (cached) return cached;
|
||||
fs.mkdirSync(path.dirname(dbPath), { recursive: true });
|
||||
const storage = new AutoresearchStorage(dbPath, projectDir);
|
||||
storageCache.set(dbPath, storage);
|
||||
return storage;
|
||||
}
|
||||
|
||||
export async function openAutoresearchStorageIfExists(cwd: string): Promise<AutoresearchStorage | null> {
|
||||
const { dbPath, projectDir } = await resolveAutoresearchPaths(cwd);
|
||||
const cached = storageCache.get(dbPath);
|
||||
if (cached) return cached;
|
||||
if (!fs.existsSync(dbPath)) return null;
|
||||
const storage = new AutoresearchStorage(dbPath, projectDir);
|
||||
storageCache.set(dbPath, storage);
|
||||
return storage;
|
||||
}
|
||||
|
||||
async function resolveAutoresearchPaths(cwd: string): Promise<{ dbPath: string; projectDir: string }> {
|
||||
const override = process.env.OMP_AUTORESEARCH_DB_DIR;
|
||||
const repoRoot = (await git.repo.root(cwd)) ?? cwd;
|
||||
const encoded = getEncodedProjectName(repoRoot);
|
||||
if (override) {
|
||||
return {
|
||||
dbPath: path.join(override, `${encoded}.db`),
|
||||
projectDir: path.join(override, encoded),
|
||||
};
|
||||
}
|
||||
return {
|
||||
dbPath: getAutoresearchDbPath(encoded),
|
||||
projectDir: getAutoresearchProjectDir(encoded),
|
||||
};
|
||||
}
|
||||
|
||||
export function closeAllAutoresearchStorages(): void {
|
||||
for (const storage of storageCache.values()) {
|
||||
try {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import * as path from "node:path";
|
||||
import { StringEnum } from "@oh-my-pi/pi-ai";
|
||||
import { Text } from "@oh-my-pi/pi-tui";
|
||||
import { Type } from "@sinclair/typebox";
|
||||
@@ -5,11 +6,16 @@ import type { ToolDefinition } from "../../extensibility/extensions";
|
||||
import type { Theme } from "../../modes/theme/theme";
|
||||
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
|
||||
import * as git from "../../utils/git";
|
||||
import { parseWorkDirDirtyPaths } from "../git";
|
||||
import { dedupeStrings, normalizePathSpec } from "../helpers";
|
||||
import { buildExperimentState } from "../state";
|
||||
import { openAutoresearchStorage, type SessionRow } from "../storage";
|
||||
import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types";
|
||||
|
||||
export const HARNESS_FILENAME = "autoresearch.sh";
|
||||
export const DEFAULT_HARNESS_COMMAND = `bash ${HARNESS_FILENAME}`;
|
||||
const HARNESS_COMMIT_TITLE = "autoresearch: harness setup";
|
||||
|
||||
const initExperimentSchema = Type.Object({
|
||||
name: Type.String({ description: "Human-readable experiment name." }),
|
||||
goal: Type.Optional(Type.String({ description: "Free-form description of what this session optimizes." })),
|
||||
@@ -23,12 +29,6 @@ const initExperimentSchema = Type.Object({
|
||||
direction: Type.Optional(
|
||||
StringEnum(["lower", "higher"], { description: "Whether lower or higher values are better. Defaults to lower." }),
|
||||
),
|
||||
preferred_command: Type.Optional(
|
||||
Type.String({
|
||||
description:
|
||||
"Preferred benchmark command for this segment. Advisory; run_experiment accepts any command but warns when the command differs.",
|
||||
}),
|
||||
),
|
||||
secondary_metrics: Type.Optional(
|
||||
Type.Array(Type.String(), {
|
||||
description: "Names of secondary metrics tracked alongside the primary metric.",
|
||||
@@ -63,6 +63,8 @@ interface InitExperimentDetails {
|
||||
createdSession: boolean;
|
||||
bumpedSegment: boolean;
|
||||
abandonedRuns: number;
|
||||
harnessCommitted: boolean;
|
||||
baselineCommit: string | null;
|
||||
}
|
||||
|
||||
export function createInitExperimentTool(
|
||||
@@ -72,7 +74,7 @@ export function createInitExperimentTool(
|
||||
name: "init_experiment",
|
||||
label: "Init Experiment",
|
||||
description:
|
||||
"Initialize or reconfigure the autoresearch session. Pass `new_segment: true` to start a fresh baseline within an existing session.",
|
||||
"Initialize or reconfigure the autoresearch session. On first call (Phase 1 → Phase 2 transition), requires `./autoresearch.sh` to exist and pending harness changes are auto-committed on an autoresearch branch. Pass `new_segment: true` to start a fresh baseline within an existing session.",
|
||||
parameters: initExperimentSchema,
|
||||
defaultInactive: true,
|
||||
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
||||
@@ -85,29 +87,63 @@ export function createInitExperimentTool(
|
||||
const offLimits = dedupeStrings((params.off_limits ?? []).map(normalizePathSpec));
|
||||
const constraints = dedupeStrings(params.constraints ?? []);
|
||||
const secondaryMetrics = dedupeStrings(params.secondary_metrics ?? []);
|
||||
const preferredCommand = params.preferred_command?.trim() || null;
|
||||
const goal = params.goal?.trim() || null;
|
||||
const maxIterations =
|
||||
params.max_iterations !== undefined && Number.isFinite(params.max_iterations) && params.max_iterations > 0
|
||||
? Math.floor(params.max_iterations)
|
||||
: null;
|
||||
const branch = (await git.branch.current(ctx.cwd)) ?? null;
|
||||
const onAutoresearchBranch = branch?.startsWith("autoresearch/") ?? false;
|
||||
|
||||
const existing = storage.getActiveSessionForBranch(branch);
|
||||
const isNewSegmentInit = existing !== null && params.new_segment === true;
|
||||
const requiresHarness = !existing || isNewSegmentInit;
|
||||
|
||||
if (requiresHarness) {
|
||||
const harnessExists = await Bun.file(path.join(ctx.cwd, HARNESS_FILENAME)).exists();
|
||||
if (!harnessExists) {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: `Error: ./${HARNESS_FILENAME} does not exist. Phase 1 of autoresearch is harness setup — write \`./${HARNESS_FILENAME}\` so it exits 0 and prints \`METRIC <name>=<value>\`, validate it via \`bash ${HARNESS_FILENAME}\`, then call init_experiment again.`,
|
||||
},
|
||||
],
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
let harnessCommitted = false;
|
||||
let commitWarning: string | null = null;
|
||||
if (requiresHarness && onAutoresearchBranch) {
|
||||
const dirty = await detectPendingChanges(ctx.cwd);
|
||||
if (dirty) {
|
||||
try {
|
||||
await git.stage.files(ctx.cwd, []);
|
||||
const message = buildHarnessCommitMessage(goal, params.name);
|
||||
await git.commit(ctx.cwd, message);
|
||||
harnessCommitted = true;
|
||||
} catch (err) {
|
||||
commitWarning = `Failed to auto-commit harness changes: ${err instanceof Error ? err.message : String(err)}. Recording baseline at current HEAD; discard may not preserve uncommitted harness files.`;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const baselineCommit = await tryReadHeadSha(ctx.cwd);
|
||||
|
||||
const existing = storage.getActiveSession();
|
||||
let session: SessionRow;
|
||||
let createdSession = false;
|
||||
let bumpedSegment = false;
|
||||
let abandonedRuns = 0;
|
||||
|
||||
if (!existing) {
|
||||
const baselineCommit = await tryReadHeadSha(ctx.cwd);
|
||||
session = storage.openSession({
|
||||
name: params.name,
|
||||
goal,
|
||||
primaryMetric: params.primary_metric,
|
||||
metricUnit,
|
||||
direction,
|
||||
preferredCommand,
|
||||
preferredCommand: DEFAULT_HARNESS_COMMAND,
|
||||
branch,
|
||||
baselineCommit,
|
||||
maxIterations,
|
||||
@@ -119,9 +155,8 @@ export function createInitExperimentTool(
|
||||
createdSession = true;
|
||||
} else {
|
||||
abandonedRuns = storage.abandonPendingRuns(existing.id);
|
||||
const updates = {
|
||||
const updates: Parameters<typeof storage.updateSession>[1] = {
|
||||
goal,
|
||||
preferredCommand,
|
||||
maxIterations,
|
||||
scopePaths,
|
||||
offLimits,
|
||||
@@ -132,8 +167,11 @@ export function createInitExperimentTool(
|
||||
direction,
|
||||
branch,
|
||||
};
|
||||
if (isNewSegmentInit) {
|
||||
updates.baselineCommit = baselineCommit;
|
||||
}
|
||||
let updated = storage.updateSession(existing.id, updates);
|
||||
if (params.new_segment === true) {
|
||||
if (isNewSegmentInit) {
|
||||
updated = storage.bumpSegment(existing.id);
|
||||
bumpedSegment = true;
|
||||
}
|
||||
@@ -159,6 +197,12 @@ export function createInitExperimentTool(
|
||||
if (abandonedRuns > 0) {
|
||||
lines.push(`Abandoned ${abandonedRuns} pending run${abandonedRuns === 1 ? "" : "s"} before reconfiguring.`);
|
||||
}
|
||||
if (harnessCommitted && session.baselineCommit) {
|
||||
lines.push(`Committed harness setup at ${session.baselineCommit.slice(0, 12)}.`);
|
||||
}
|
||||
if (commitWarning) {
|
||||
lines.push(commitWarning);
|
||||
}
|
||||
if (createdSession) {
|
||||
lines.push(`Started session #${session.id}: ${session.name}`);
|
||||
} else if (bumpedSegment) {
|
||||
@@ -169,9 +213,7 @@ export function createInitExperimentTool(
|
||||
lines.push(
|
||||
`Metric: ${session.primaryMetric} (${session.metricUnit || "unitless"}, ${session.direction} is better)`,
|
||||
);
|
||||
if (session.preferredCommand) {
|
||||
lines.push(`Preferred command: ${session.preferredCommand}`);
|
||||
}
|
||||
lines.push(`Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`);
|
||||
if (session.scopePaths.length > 0) {
|
||||
lines.push(`Files in scope: ${session.scopePaths.join(", ")}`);
|
||||
}
|
||||
@@ -188,10 +230,17 @@ export function createInitExperimentTool(
|
||||
lines.push(`Baseline commit: ${session.baselineCommit.slice(0, 12)}`);
|
||||
}
|
||||
if (createdSession) {
|
||||
lines.push("Run the baseline experiment now and log it.");
|
||||
lines.push(
|
||||
"Phase 2: iteration loop is active. Run the baseline experiment with `run_experiment` and log it.",
|
||||
);
|
||||
} else if (bumpedSegment) {
|
||||
lines.push("Run a fresh baseline for the new segment.");
|
||||
}
|
||||
if (requiresHarness && !onAutoresearchBranch) {
|
||||
lines.push(
|
||||
"Note: not on a dedicated `autoresearch/*` branch — `log_experiment discard` will only revert run-modified files, not reset to baseline.",
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
content: [{ type: "text", text: lines.join("\n") }],
|
||||
@@ -200,6 +249,8 @@ export function createInitExperimentTool(
|
||||
createdSession,
|
||||
bumpedSegment,
|
||||
abandonedRuns,
|
||||
harnessCommitted,
|
||||
baselineCommit: session.baselineCommit,
|
||||
},
|
||||
};
|
||||
},
|
||||
@@ -224,3 +275,23 @@ async function tryReadHeadSha(cwd: string): Promise<string | null> {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function detectPendingChanges(cwd: string): Promise<boolean> {
|
||||
try {
|
||||
const statusText = await git.status(cwd, { porcelainV1: true, untrackedFiles: "all", z: true });
|
||||
const workDirPrefix = await git.show.prefix(cwd).catch(() => "");
|
||||
return parseWorkDirDirtyPaths(statusText, workDirPrefix).length > 0;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function buildHarnessCommitMessage(goal: string | null, name: string): string {
|
||||
const lines = [HARNESS_COMMIT_TITLE, "", `Benchmark entrypoint: ${DEFAULT_HARNESS_COMMAND}`];
|
||||
if (goal) {
|
||||
lines.push(`Goal: ${goal}`);
|
||||
} else {
|
||||
lines.push(`Session: ${name}`);
|
||||
}
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
@@ -7,7 +7,7 @@ import type { ToolDefinition } from "../../extensibility/extensions";
|
||||
import type { Theme } from "../../modes/theme/theme";
|
||||
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
|
||||
import * as git from "../../utils/git";
|
||||
import { computeRunModifiedPaths, getCurrentAutoresearchBranch } from "../git";
|
||||
import { computeRunModifiedPaths, getCurrentAutoresearchBranch, parseWorkDirDirtyPaths } from "../git";
|
||||
import { ensureNumericMetricMap, formatNum, mergeAsi, pathMatchesSpec, sanitizeAsi } from "../helpers";
|
||||
import {
|
||||
buildExperimentState,
|
||||
@@ -16,7 +16,7 @@ import {
|
||||
findBaselineSecondary,
|
||||
findBestKeptMetric,
|
||||
} from "../state";
|
||||
import { openAutoresearchStorage, type SessionRow } from "../storage";
|
||||
import { openAutoresearchStorageIfExists, type SessionRow } from "../storage";
|
||||
import type {
|
||||
ASIData,
|
||||
AutoresearchToolFactoryOptions,
|
||||
@@ -81,14 +81,15 @@ export function createLogExperimentTool(
|
||||
parameters: logExperimentSchema,
|
||||
defaultInactive: true,
|
||||
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
if (!session) {
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
|
||||
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
|
||||
if (!storage || !session) {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Error: no active autoresearch session. Call init_experiment first.",
|
||||
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
|
||||
},
|
||||
],
|
||||
};
|
||||
@@ -113,8 +114,23 @@ export function createLogExperimentTool(
|
||||
const branchName = await getCurrentAutoresearchBranch(options.pi, ctx.cwd);
|
||||
const onAutoresearchBranch = branchName !== null;
|
||||
|
||||
const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths(ctx.cwd, pendingRun.preRunDirtyPaths);
|
||||
const allModified = [...modifiedTracked, ...modifiedUntracked];
|
||||
let allModified: string[];
|
||||
if (onAutoresearchBranch) {
|
||||
// On a dedicated autoresearch branch every iteration starts from a clean
|
||||
// worktree (init_experiment baseline + previous keep commit / discard reset),
|
||||
// so any currently-dirty path is the agent's iteration change. Off-branch we
|
||||
// can't tell user dirt apart from agent edits, so we keep the (lossy)
|
||||
// preRunDirtyPaths filter.
|
||||
const statusText = await tryGitStatus(ctx.cwd);
|
||||
const workDirPrefix = await tryGitPrefix(ctx.cwd);
|
||||
allModified = parseWorkDirDirtyPaths(statusText, workDirPrefix);
|
||||
} else {
|
||||
const { modifiedTracked, modifiedUntracked } = await detectModifiedPaths(
|
||||
ctx.cwd,
|
||||
pendingRun.preRunDirtyPaths,
|
||||
);
|
||||
allModified = [...modifiedTracked, ...modifiedUntracked];
|
||||
}
|
||||
const scopeDeviations = computeScopeDeviations(allModified, session);
|
||||
|
||||
const justification = params.justification?.trim() || null;
|
||||
@@ -165,7 +181,6 @@ export function createLogExperimentTool(
|
||||
ctx.cwd,
|
||||
pendingRun.preRunDirtyPaths,
|
||||
onAutoresearchBranch,
|
||||
session.baselineCommit,
|
||||
);
|
||||
if (revertResult.error) {
|
||||
return {
|
||||
@@ -346,14 +361,15 @@ async function revertFailedExperiment(
|
||||
cwd: string,
|
||||
preRunDirtyPaths: string[],
|
||||
onAutoresearchBranch: boolean,
|
||||
baselineCommit: string | null,
|
||||
): Promise<KeepCommitResult> {
|
||||
if (onAutoresearchBranch) {
|
||||
// Discard reverts only the current iteration's uncommitted changes — never
|
||||
// rewinds prior `keep` commits. Reset to HEAD so any kept improvements
|
||||
// already on the branch survive.
|
||||
try {
|
||||
const target = baselineCommit && baselineCommit.length > 0 ? baselineCommit : "HEAD";
|
||||
await git.reset(cwd, { hard: true, target });
|
||||
await git.reset(cwd, { hard: true, target: "HEAD" });
|
||||
await git.clean(cwd);
|
||||
return { note: `worktree reset to ${target.slice(0, 12)}` };
|
||||
return { note: "worktree reset to HEAD" };
|
||||
} catch (err) {
|
||||
return { error: `git reset/clean failed: ${err instanceof Error ? err.message : String(err)}` };
|
||||
}
|
||||
|
||||
@@ -7,7 +7,7 @@ import { Type } from "@sinclair/typebox";
|
||||
import type { ToolDefinition } from "../../extensibility/extensions";
|
||||
import type { Theme } from "../../modes/theme/theme";
|
||||
import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output";
|
||||
import { replaceTabs, shortenPath, truncateToWidth } from "../../tools/render-utils";
|
||||
import { replaceTabs, shortenPath } from "../../tools/render-utils";
|
||||
import * as git from "../../utils/git";
|
||||
import { parseWorkDirDirtyPaths } from "../git";
|
||||
import {
|
||||
@@ -20,11 +20,11 @@ import {
|
||||
parseMetricLines,
|
||||
} from "../helpers";
|
||||
import { buildExperimentState } from "../state";
|
||||
import { openAutoresearchStorage } from "../storage";
|
||||
import { openAutoresearchStorageIfExists } from "../storage";
|
||||
import type { AutoresearchToolFactoryOptions, RunDetails, RunExperimentProgressDetails } from "../types";
|
||||
import { DEFAULT_HARNESS_COMMAND } from "./init-experiment";
|
||||
|
||||
const runExperimentSchema = Type.Object({
|
||||
command: Type.String({ description: "Shell command to run for this experiment." }),
|
||||
timeout_seconds: Type.Optional(Type.Number({ description: "Timeout in seconds. Defaults to 600." })),
|
||||
});
|
||||
|
||||
@@ -54,14 +54,15 @@ export function createRunExperimentTool(
|
||||
parameters: runExperimentSchema,
|
||||
defaultInactive: true,
|
||||
async execute(_toolCallId, params, signal, onUpdate, ctx) {
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
if (!session) {
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
|
||||
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
|
||||
if (!storage || !session) {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Error: no active autoresearch session. Call init_experiment first.",
|
||||
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
|
||||
},
|
||||
],
|
||||
};
|
||||
@@ -76,11 +77,7 @@ export function createRunExperimentTool(
|
||||
return pending.id;
|
||||
})();
|
||||
|
||||
let commandWarning: string | null = null;
|
||||
if (session.preferredCommand && params.command.trim() !== session.preferredCommand.trim()) {
|
||||
commandWarning = `Note: command differs from preferred (\`${session.preferredCommand}\`). Re-init the experiment if the workload itself changed.`;
|
||||
}
|
||||
|
||||
const resolvedCommand = DEFAULT_HARNESS_COMMAND;
|
||||
const preRunStatus = await tryGitStatus(ctx.cwd);
|
||||
const workDirPrefix = await tryGitPrefix(ctx.cwd);
|
||||
const preRunDirtyPaths = parseWorkDirDirtyPaths(preRunStatus, workDirPrefix);
|
||||
@@ -89,7 +86,7 @@ export function createRunExperimentTool(
|
||||
const insertedRun = storage.insertRun({
|
||||
sessionId: session.id,
|
||||
segment: session.currentSegment,
|
||||
command: params.command,
|
||||
command: resolvedCommand,
|
||||
logPath: "", // patched after we know the run id
|
||||
preRunDirtyPaths,
|
||||
startedAt,
|
||||
@@ -107,7 +104,7 @@ export function createRunExperimentTool(
|
||||
runtime.lastRunSummary = null;
|
||||
runtime.runningExperiment = {
|
||||
startedAt,
|
||||
command: params.command,
|
||||
command: resolvedCommand,
|
||||
runDirectory,
|
||||
runNumber: insertedRun.id,
|
||||
};
|
||||
@@ -118,7 +115,7 @@ export function createRunExperimentTool(
|
||||
let execution: ProcessExecutionResult;
|
||||
try {
|
||||
execution = await executeProcess({
|
||||
command: ["bash", "-lc", params.command],
|
||||
command: ["bash", "-lc", resolvedCommand],
|
||||
cwd: ctx.cwd,
|
||||
logPath: benchmarkLogPath,
|
||||
timeoutMs,
|
||||
@@ -178,7 +175,7 @@ export function createRunExperimentTool(
|
||||
runNumber: insertedRun.id,
|
||||
runDirectory,
|
||||
benchmarkLogPath,
|
||||
command: params.command,
|
||||
command: resolvedCommand,
|
||||
exitCode: execution.exitCode,
|
||||
durationSeconds,
|
||||
passed,
|
||||
@@ -191,14 +188,13 @@ export function createRunExperimentTool(
|
||||
metricName: session.primaryMetric,
|
||||
metricUnit: session.metricUnit,
|
||||
preRunDirtyPaths,
|
||||
commandWarning,
|
||||
abandonedPriorRun,
|
||||
truncation: llmTruncation.truncated ? llmTruncation : undefined,
|
||||
fullOutputPath: execution.logPath,
|
||||
};
|
||||
|
||||
runtime.lastRunSummary = {
|
||||
command: params.command,
|
||||
command: resolvedCommand,
|
||||
durationSeconds,
|
||||
parsedAsi,
|
||||
parsedMetrics,
|
||||
@@ -222,7 +218,6 @@ export function createRunExperimentTool(
|
||||
options.dashboard.requestRender();
|
||||
|
||||
const headerLines: string[] = [];
|
||||
if (commandWarning) headerLines.push(commandWarning);
|
||||
if (abandonedPriorRun !== null) {
|
||||
headerLines.push(`Note: abandoned prior pending run #${abandonedPriorRun} before starting this run.`);
|
||||
}
|
||||
@@ -238,10 +233,9 @@ export function createRunExperimentTool(
|
||||
details: resultDetails,
|
||||
};
|
||||
},
|
||||
renderCall(args, _options, theme): Text {
|
||||
const commandPreview = truncateToWidth(replaceTabs(args.command), 100);
|
||||
renderCall(_args, _options, theme): Text {
|
||||
return new Text(
|
||||
`${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", commandPreview)}`,
|
||||
`${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", DEFAULT_HARNESS_COMMAND)}`,
|
||||
0,
|
||||
0,
|
||||
);
|
||||
|
||||
@@ -3,8 +3,9 @@ import { Type } from "@sinclair/typebox";
|
||||
import type { ToolDefinition } from "../../extensibility/extensions";
|
||||
import type { Theme } from "../../modes/theme/theme";
|
||||
import { replaceTabs, truncateToWidth } from "../../tools/render-utils";
|
||||
import * as git from "../../utils/git";
|
||||
import { buildExperimentState } from "../state";
|
||||
import { openAutoresearchStorage } from "../storage";
|
||||
import { openAutoresearchStorageIfExists } from "../storage";
|
||||
import type { AutoresearchToolFactoryOptions } from "../types";
|
||||
|
||||
const updateNotesSchema = Type.Object({
|
||||
@@ -34,14 +35,15 @@ export function createUpdateNotesTool(
|
||||
parameters: updateNotesSchema,
|
||||
defaultInactive: true,
|
||||
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
||||
const storage = await openAutoresearchStorage(ctx.cwd);
|
||||
const session = storage.getActiveSession();
|
||||
if (!session) {
|
||||
const storage = await openAutoresearchStorageIfExists(ctx.cwd);
|
||||
const currentBranch = (await git.branch.current(ctx.cwd)) ?? null;
|
||||
const session = storage?.getActiveSessionForBranch(currentBranch) ?? null;
|
||||
if (!storage || !session) {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Error: no active autoresearch session. Call init_experiment first.",
|
||||
text: "Error: no active autoresearch session for the current branch. Call init_experiment first.",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
@@ -51,7 +51,6 @@ export interface ExperimentState {
|
||||
currentSegment: number;
|
||||
maxExperiments: number | null;
|
||||
confidence: number | null;
|
||||
benchmarkCommand: string | null;
|
||||
scopePaths: string[];
|
||||
offLimits: string[];
|
||||
constraints: string[];
|
||||
@@ -86,7 +85,6 @@ export interface RunDetails {
|
||||
metricName: string;
|
||||
metricUnit: string;
|
||||
preRunDirtyPaths: string[];
|
||||
commandWarning: string | null;
|
||||
abandonedPriorRun: number | null;
|
||||
truncation?: TruncationResult;
|
||||
fullOutputPath?: string;
|
||||
|
||||
@@ -84,6 +84,10 @@ async function checkoutBranch(dir: string, name: string): Promise<void> {
|
||||
await $`git checkout -b ${name}`.cwd(dir).quiet();
|
||||
}
|
||||
|
||||
async function writeHarnessStub(dir: string, body = "echo METRIC m=1"): Promise<void> {
|
||||
await Bun.write(path.join(dir, "autoresearch.sh"), `#!/usr/bin/env bash\n${body}\n`);
|
||||
}
|
||||
|
||||
describe("init_experiment", () => {
|
||||
let dbOverride: string;
|
||||
|
||||
@@ -99,6 +103,7 @@ describe("init_experiment", () => {
|
||||
|
||||
it("opens a new session and persists scope and metric metadata", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -114,7 +119,6 @@ describe("init_experiment", () => {
|
||||
primary_metric: "runtime_ms",
|
||||
metric_unit: "ms",
|
||||
direction: "lower",
|
||||
preferred_command: "bun bench",
|
||||
scope_paths: ["src", "src/foo"],
|
||||
off_limits: ["test"],
|
||||
secondary_metrics: ["memory_mb"],
|
||||
@@ -141,6 +145,7 @@ describe("init_experiment", () => {
|
||||
|
||||
it("updates fields without bumping segment when no new_segment flag is passed", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -171,6 +176,7 @@ describe("init_experiment", () => {
|
||||
|
||||
it("bumps segment when new_segment is true on a re-init", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -188,6 +194,78 @@ describe("init_experiment", () => {
|
||||
expect(result.details?.bumpedSegment).toBe(true);
|
||||
expect(result.details?.state.currentSegment).toBe(1);
|
||||
});
|
||||
|
||||
it("rejects when autoresearch.sh is missing on first init", async () => {
|
||||
const dir = makeTempDir();
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-1",
|
||||
{ name: "x", primary_metric: "m" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
expect(firstTextBlockText(result.content)).toContain("autoresearch.sh");
|
||||
const storage = await openAutoresearchStorage(dir);
|
||||
expect(storage.getActiveSession()).toBeNull();
|
||||
});
|
||||
|
||||
it("auto-commits pending harness changes on an autoresearch branch", async () => {
|
||||
const dir = makeTempDir();
|
||||
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
|
||||
await checkoutBranch(dir, "autoresearch/setup-test");
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-1",
|
||||
{ name: "x", primary_metric: "m", goal: "speed" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
expect(result.details?.harnessCommitted).toBe(true);
|
||||
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
|
||||
expect(newHead).not.toBe(initialBaseline);
|
||||
expect(result.details?.baselineCommit).toBe(newHead);
|
||||
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
|
||||
expect(status).toBe("");
|
||||
const message = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim();
|
||||
expect(message).toContain("autoresearch: harness setup");
|
||||
});
|
||||
|
||||
it("does not auto-commit when not on an autoresearch branch", async () => {
|
||||
const dir = makeTempDir();
|
||||
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-1",
|
||||
{ name: "x", primary_metric: "m" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
expect(result.details?.harnessCommitted).toBe(false);
|
||||
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
|
||||
expect(newHead).toBe(initialBaseline);
|
||||
// Harness file is still in the worktree, untracked.
|
||||
expect(fs.existsSync(path.join(dir, "autoresearch.sh"))).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("run_experiment", () => {
|
||||
@@ -211,12 +289,13 @@ describe("run_experiment", () => {
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await run.execute("call-1", { command: "echo hi" }, undefined, undefined, createCtx(dir));
|
||||
const result = await run.execute("call-1", {}, undefined, undefined, createCtx(dir));
|
||||
expect(firstTextBlockText(result.content)).toContain("no active autoresearch session");
|
||||
});
|
||||
|
||||
it("accepts arbitrary commands, parses METRIC/ASI, and stores a run", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir, "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline");
|
||||
const runtime = createSessionRuntime();
|
||||
const init = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -235,16 +314,7 @@ describe("run_experiment", () => {
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await run.execute(
|
||||
"r",
|
||||
{
|
||||
command: "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline",
|
||||
timeout_seconds: 5,
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
const result = await run.execute("r", { timeout_seconds: 5 }, undefined, undefined, createCtx(dir));
|
||||
const details = result.details as RunDetails;
|
||||
expect(details.parsedPrimary).toBe(42);
|
||||
expect(details.parsedMetrics).toMatchObject({ runtime_ms: 42, memory_mb: 12 });
|
||||
@@ -262,6 +332,7 @@ describe("run_experiment", () => {
|
||||
|
||||
it("abandons a prior pending run instead of blocking", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const initTool = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -274,37 +345,32 @@ describe("run_experiment", () => {
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
await run.execute("r1", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir));
|
||||
const result = await run.execute("r2", { command: "echo METRIC m=2" }, undefined, undefined, createCtx(dir));
|
||||
await run.execute("r1", {}, undefined, undefined, createCtx(dir));
|
||||
const result = await run.execute("r2", {}, undefined, undefined, createCtx(dir));
|
||||
const details = result.details as RunDetails;
|
||||
expect(details.abandonedPriorRun).not.toBeNull();
|
||||
expect(details.runNumber).not.toBe(details.abandonedPriorRun);
|
||||
});
|
||||
|
||||
it("warns when command differs from the preferred command", async () => {
|
||||
it("runs ./autoresearch.sh and parses METRIC/ASI from its output", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir, "echo METRIC m=99");
|
||||
const runtime = createSessionRuntime();
|
||||
const init = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
await init.execute(
|
||||
"i",
|
||||
{ name: "x", primary_metric: "m", preferred_command: "echo preferred" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
|
||||
const run = createRunExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createPiHarness().api,
|
||||
});
|
||||
const result = await run.execute("r", { command: "echo METRIC m=1" }, undefined, undefined, createCtx(dir));
|
||||
const result = await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
const details = result.details as RunDetails;
|
||||
expect(details.commandWarning).toContain("preferred");
|
||||
expect(firstTextBlockText(result.content)).toContain("preferred");
|
||||
expect(details.command).toBe("bash autoresearch.sh");
|
||||
expect(details.parsedPrimary).toBe(99);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -322,6 +388,7 @@ describe("log_experiment", () => {
|
||||
});
|
||||
|
||||
async function setupRun(dir: string, runtime = createSessionRuntime()) {
|
||||
await writeHarnessStub(dir, "echo METRIC runtime_ms=10");
|
||||
const harness = createPiHarness();
|
||||
const init = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -346,7 +413,7 @@ describe("log_experiment", () => {
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await run.execute("r", { command: "echo METRIC runtime_ms=10" }, undefined, undefined, createCtx(dir));
|
||||
await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
const log = createLogExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
@@ -357,6 +424,7 @@ describe("log_experiment", () => {
|
||||
|
||||
it("rejects when no pending run exists", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const harness = createPiHarness();
|
||||
const init = createInitExperimentTool({
|
||||
@@ -480,7 +548,7 @@ describe("log_experiment", () => {
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await run.execute("r2", { command: "echo METRIC runtime_ms=8" }, undefined, undefined, createCtx(dir));
|
||||
await run.execute("r2", {}, undefined, undefined, createCtx(dir));
|
||||
const log2 = createLogExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
@@ -512,6 +580,7 @@ describe("log_experiment", () => {
|
||||
|
||||
it("on a non-autoresearch branch, discard reverts only run-modified files", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
await initGitRepo(dir);
|
||||
// Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths.
|
||||
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
|
||||
@@ -539,7 +608,7 @@ describe("log_experiment", () => {
|
||||
});
|
||||
// Pre-existing untracked file (will not be touched by revert because it was dirty before run)
|
||||
await Bun.write(path.join(dir, "preexisting.txt"), "leave me\n");
|
||||
await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir));
|
||||
await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
// Simulate a run-introduced change
|
||||
await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 2;\n");
|
||||
await Bun.write(path.join(dir, "src", "new.ts"), "export const NEW = true;\n");
|
||||
@@ -564,9 +633,13 @@ describe("log_experiment", () => {
|
||||
expect(fs.readFileSync(path.join(dir, "src", "edit-me.ts"), "utf8")).toBe("export const v = 1;\n");
|
||||
});
|
||||
|
||||
it("on an autoresearch branch, discard resets the worktree to baseline_commit", async () => {
|
||||
it("on an autoresearch branch, discard reverts uncommitted changes but preserves prior commits", async () => {
|
||||
const dir = makeTempDir();
|
||||
const { baselineCommit } = await initGitRepo(dir);
|
||||
await initGitRepo(dir);
|
||||
// Commit the harness on main so it is part of the autoresearch branch's baseline.
|
||||
await writeHarnessStub(dir);
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m harness`.cwd(dir).quiet();
|
||||
await checkoutBranch(dir, "autoresearch/test-20260501");
|
||||
const runtime = createSessionRuntime();
|
||||
const harness = createPiHarness();
|
||||
@@ -576,16 +649,21 @@ describe("log_experiment", () => {
|
||||
pi: harness.api,
|
||||
});
|
||||
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
|
||||
// Simulate a previously kept iteration by committing it directly on the branch.
|
||||
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n");
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m "kept iteration"`.cwd(dir).quiet();
|
||||
const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
|
||||
|
||||
const run = createRunExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await run.execute("r", { command: "echo METRIC m=10" }, undefined, undefined, createCtx(dir));
|
||||
// Modify and commit a file on the autoresearch branch — discard should reset HEAD back to baseline.
|
||||
await Bun.write(path.join(dir, "src", "stub.ts"), "export const v = 1;\n");
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m wip`.cwd(dir).quiet();
|
||||
await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
// Current iteration's uncommitted edits.
|
||||
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 999;\n");
|
||||
await Bun.write(path.join(dir, "scratch.ts"), "// junk\n");
|
||||
|
||||
const log = createLogExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
@@ -599,9 +677,117 @@ describe("log_experiment", () => {
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
const headSha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
|
||||
expect(headSha).toBe(baselineCommit);
|
||||
expect(fs.existsSync(path.join(dir, "src", "stub.ts"))).toBe(false);
|
||||
const headAfter = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
|
||||
// Prior commits survive — discard does not rewind history.
|
||||
expect(headAfter).toBe(headBeforeDiscard);
|
||||
// Uncommitted iteration changes are gone.
|
||||
expect(fs.readFileSync(path.join(dir, "src", "kept.ts"), "utf8")).toBe("export const v = 1;\n");
|
||||
expect(fs.existsSync(path.join(dir, "scratch.ts"))).toBe(false);
|
||||
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
|
||||
expect(status).toBe("");
|
||||
});
|
||||
|
||||
it("on an autoresearch branch, keep commits files that were dirty before run_experiment", async () => {
|
||||
const dir = makeTempDir();
|
||||
await initGitRepo(dir);
|
||||
await writeHarnessStub(dir);
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m harness`.cwd(dir).quiet();
|
||||
// Seed a tracked file that the agent will edit during the iteration.
|
||||
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
|
||||
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n");
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m seed`.cwd(dir).quiet();
|
||||
await checkoutBranch(dir, "autoresearch/keep-test");
|
||||
const runtime = createSessionRuntime();
|
||||
const harness = createPiHarness();
|
||||
const init = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await init.execute(
|
||||
"i",
|
||||
{ name: "x", primary_metric: "m", scope_paths: ["src"] },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
// Agent edits BEFORE running the benchmark — the iteration's diff is dirty
|
||||
// at run_experiment time.
|
||||
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 2;\n");
|
||||
const run = createRunExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
|
||||
const log = createLogExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
const result = await log.execute(
|
||||
"l",
|
||||
{ metric: 42, status: "keep", description: "improvement" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
const details = result.details as LogDetails;
|
||||
expect(details.experiment.modifiedPaths).toContain("src/store.ts");
|
||||
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
|
||||
expect(status).toBe("");
|
||||
const lastMsg = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim();
|
||||
expect(lastMsg).toContain("improvement");
|
||||
});
|
||||
|
||||
it("flags off-scope dirty files even when they were dirty before run_experiment", async () => {
|
||||
const dir = makeTempDir();
|
||||
await initGitRepo(dir);
|
||||
await writeHarnessStub(dir);
|
||||
await $`git add -A`.cwd(dir).quiet();
|
||||
await $`git commit -m harness`.cwd(dir).quiet();
|
||||
await checkoutBranch(dir, "autoresearch/scope-test");
|
||||
const runtime = createSessionRuntime();
|
||||
const harness = createPiHarness();
|
||||
const init = createInitExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await init.execute(
|
||||
"i",
|
||||
{ name: "x", primary_metric: "m", scope_paths: ["src"], off_limits: ["forbidden"] },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
// Off-scope edit BEFORE run_experiment.
|
||||
fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true });
|
||||
await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n");
|
||||
const run = createRunExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
await run.execute("r", {}, undefined, undefined, createCtx(dir));
|
||||
|
||||
const log = createLogExperimentTool({
|
||||
dashboard: dashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: harness.api,
|
||||
});
|
||||
const result = await log.execute(
|
||||
"l",
|
||||
{ metric: 42, status: "keep", description: "off-scope" },
|
||||
undefined,
|
||||
undefined,
|
||||
createCtx(dir),
|
||||
);
|
||||
const details = result.details as LogDetails;
|
||||
expect(details.scopeDeviations).toContain("forbidden/x.ts");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -620,6 +806,7 @@ describe("update_notes", () => {
|
||||
|
||||
it("replaces session notes and refreshes runtime state", async () => {
|
||||
const dir = makeTempDir();
|
||||
await writeHarnessStub(dir);
|
||||
const runtime = createSessionRuntime();
|
||||
const harness = createPiHarness();
|
||||
const init = createInitExperimentTool({
|
||||
|
||||
Reference in New Issue
Block a user