diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 70e80375f..8d165f70b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -14,6 +14,21 @@ ### Added +- Added autoresearch contract system for validating benchmark commands, metrics, scope paths, off-limits paths, and constraints with fingerprint tracking to detect configuration drift +- Added `autoresearch.program.md` support for repo-local playbook overlays that guide session strategy while preserving `autoresearch.md` as source of truth +- Added pending run artifact tracking and recovery to resume incomplete experiments from `.autoresearch/runs/` directory with run numbers and benchmark logs +- Added run directory organization with numbered run artifacts, benchmark logs, and optional checks logs for experiment traceability +- Added segment fingerprinting to detect when benchmark configuration changes between runs and warn about potential incomparability +- Added support for secondary metrics tracking alongside primary metric with configurable direction (lower/higher is better) +- Added `getCurrentAutoresearchBranch()` helper to detect and validate existing autoresearch branches for session resumption +- Added `PendingRunSummary` type to track unlogged run state including parsed metrics, ASI data, and pass/fail status +- Added hidden next-turn message delivery via `deliverAs: 'nextTurn'` with optional `triggerTurn` to queue context for next LLM call without exposing in editable queue +- Added `#queueHiddenNextTurnMessage()` and `#promptQueuedHiddenNextTurnMessages()` to AgentSession for autonomous tool reactions +- Added resume context support in `command-resume.md` template for user-provided guidance when resuming sessions +- Added current segment snapshot display in autoresearch prompt showing recent runs, baseline metrics, and best results +- Added pending run indicator in autoresearch prompt to guide users to complete unlogged experiments before starting new benchmarks +- Added local playbook section in autoresearch prompt when `autoresearch.program.md` exists +- Added tab replacement in dashboard and tool output rendering to prevent display corruption from shell commands with tabs - Added boundary duplication warning when replace_range or replace_line operations include a last inserted line that matches the next surviving line, helping detect off-by-one range errors - Added git branch isolation for autoresearch sessions via `ensureAutoresearchBranch()` to safely revert failed experiments - Added branch status line to autoresearch initialization and resume prompts showing created or reused branch name @@ -47,6 +62,20 @@ ### Changed +- Changed autoresearch initialization to collect and validate benchmark command, metric definition, scope paths, off-limits list, and constraints before `init_experiment` +- Changed `init_experiment` to require exact benchmark command, metric definition, scope, off-limits, and constraints matching collected contract +- Changed `log_experiment` to record run number, benchmark command, scope paths, off-limits list, constraints, and segment fingerprint with each result +- Changed `run_experiment` to organize output in numbered run directories with separate benchmark and checks logs for artifact preservation +- Changed autoresearch dashboard to show pending run indicator when unlogged experiment exists +- Changed autoresearch resume workflow to detect and offer recovery of pending run artifacts before continuing experiment loop +- Changed `ExperimentResult` to include `runNumber`, `benchmarkCommand`, `scopePaths`, `offLimits`, `constraints`, and `segmentFingerprint` fields +- Changed `RunningExperiment` to track `runDirectory` and `runNumber` for artifact organization +- Changed `AutoresearchRuntime` to include `lastRunArtifactDir`, `lastRunNumber`, `lastRunSummary`, `benchmarkCommand`, `secondaryMetrics`, `scopePaths`, `offLimits`, `constraints`, and `segmentFingerprint` +- Changed autoresearch prompts to emphasize `autoresearch.md` as source of truth for benchmark, scope, and constraints +- Changed `command-initialize.md` to display collected setup (benchmark command, metric, direction, scope, off-limits, constraints) before initialization +- Changed `resume-message.md` to reference pending run artifacts and guide completion of unlogged experiments +- Changed `sendMessage()` API documentation to clarify `deliverAs: 'nextTurn'` behavior for hidden context delivery +- Changed `SendMessageHandler` type documentation to explain hidden next-turn message queuing during prompt teardown - Changed autoresearch startup to create or reuse a dedicated `autoresearch/...` git branch before enabling the experiment loop - Changed autoresearch to refuse startup when unrelated worktree changes would make auto-reverts unsafe - Changed autoresearch prompts to emphasize scope and constraints as source of truth for session direction @@ -82,6 +111,9 @@ ### Fixed +- Fixed autoresearch resume to detect and recover pending run artifacts that were left unlogged from previous sessions +- Fixed dashboard overlay to display when running experiment even with zero completed results +- Fixed tab character rendering in dashboard command display and tool output summaries - Fixed autoresearch logging to require durable ASI metadata (hypothesis, rollback_reason, next_action_hint) for every run including rollback context for discarded, crashed, and checks-failed experiments - Fixed autoresearch logging to require durable ASI metadata for every run, including rollback context for discarded, crashed, and checks-failed experiments diff --git a/packages/coding-agent/src/autoresearch/command-initialize.md b/packages/coding-agent/src/autoresearch/command-initialize.md index 271d9135a..9986a844b 100644 --- a/packages/coding-agent/src/autoresearch/command-initialize.md +++ b/packages/coding-agent/src/autoresearch/command-initialize.md @@ -4,13 +4,27 @@ Set up autoresearch for this intent: {{branch_status_line}} +Collected setup: + +- benchmark command: `{{benchmark_command}}` +- primary metric: `{{metric_name}}` +- metric unit: `{{metric_unit}}` +- direction: `{{direction}}` +- files in scope: +{{{scope_paths_block}}} +- off limits: +{{{off_limits_block}}} +- constraints: +{{{constraints_block}}} + Explain briefly what autoresearch will do in this repository, then initialize the workspace. Your first actions: - write `autoresearch.md` -- define `Files in Scope`, `Off Limits`, and `Constraints` in `autoresearch.md` +- record the collected benchmark command, primary metric, metric unit, direction, scope, off-limits list, and constraints in `autoresearch.md` +- optionally write `autoresearch.program.md` when a repo-local playbook would help future resume quality - define the benchmark entrypoint in `autoresearch.sh` - optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate -- run `init_experiment` +- run `init_experiment` with the exact collected benchmark command, metric definition, scope paths, off-limits list, and constraints - run and log the baseline - keep iterating until interrupted or until the configured iteration cap is reached diff --git a/packages/coding-agent/src/autoresearch/command-resume.md b/packages/coding-agent/src/autoresearch/command-resume.md index 46d8cc742..3dd0030a4 100644 --- a/packages/coding-agent/src/autoresearch/command-resume.md +++ b/packages/coding-agent/src/autoresearch/command-resume.md @@ -3,6 +3,12 @@ Resume autoresearch from the attached notes. @{{autoresearch_md_path}} {{branch_status_line}} +{{#if has_resume_context}} + +Additional context from the user: + +{{resume_context}} +{{/if}} Use the notes as the source of truth for the current direction, scope, and constraints. - inspect recent git history for context diff --git a/packages/coding-agent/src/autoresearch/contract.ts b/packages/coding-agent/src/autoresearch/contract.ts new file mode 100644 index 000000000..45d4c4e46 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/contract.ts @@ -0,0 +1,318 @@ +import * as crypto from "node:crypto"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import type { AutoresearchBenchmarkContract, AutoresearchContract, MetricDirection } from "./types"; + +export interface AutoresearchContractLoadResult { + contract: AutoresearchContract; + errors: string[]; + path: string; +} + +export interface AutoresearchScriptSnapshot { + benchmarkScript: string; + benchmarkScriptPath: string; + checksScript: string | null; + checksScriptPath: string; + errors: string[]; +} + +const HEADING_REGEX = /^##\s+(.+?)\s*$/; +const LIST_ITEM_REGEX = /^\s*[-*]\s+(.*)$/; +const KEY_VALUE_REGEX = /^\s*[-*]\s+([^:]+):\s*(.*)$/; + +export function readAutoresearchContract(workDir: string): AutoresearchContractLoadResult { + const contractPath = path.join(workDir, "autoresearch.md"); + let content = ""; + try { + content = fs.readFileSync(contractPath, "utf8"); + } catch { + return { + contract: createEmptyAutoresearchContract(), + errors: [`${contractPath} does not exist. Create it before initializing autoresearch.`], + path: contractPath, + }; + } + + const contract = parseAutoresearchContract(content); + const errors = validateAutoresearchContract(contract); + return { contract, errors, path: contractPath }; +} + +export function parseAutoresearchContract(markdown: string): AutoresearchContract { + const sections = extractSections(markdown); + return { + benchmark: parseBenchmarkSection(sections.get("benchmark") ?? ""), + scopePaths: parseListSection(sections.get("files in scope") ?? "", normalizeContractPathSpec), + offLimits: parseListSection(sections.get("off limits") ?? "", normalizeContractPathSpec), + constraints: parseListSection(sections.get("constraints") ?? ""), + }; +} + +export function validateAutoresearchContract(contract: AutoresearchContract): string[] { + const errors: string[] = []; + if (!contract.benchmark.command) { + errors.push("Benchmark.command is required in autoresearch.md."); + } + if (!contract.benchmark.primaryMetric) { + errors.push("Benchmark.primary metric is required in autoresearch.md."); + } + if (!contract.benchmark.direction) { + errors.push("Benchmark.direction must be `lower` or `higher` in autoresearch.md."); + } + if (contract.scopePaths.length === 0) { + errors.push("Files in Scope must contain at least one path in autoresearch.md."); + } + return errors; +} + +export function buildAutoresearchSegmentFingerprint( + contract: AutoresearchContract, + scripts: { + benchmarkScript: string; + checksScript: string | null; + }, +): string { + const payload = { + benchmark: contract.benchmark, + scopePaths: contract.scopePaths, + offLimits: contract.offLimits, + constraints: contract.constraints, + scripts, + }; + return crypto.createHash("sha256").update(JSON.stringify(payload)).digest("hex"); +} + +export function getAutoresearchFingerprintMismatchError( + stateFingerprint: string | null, + workDir: string, +): string | null { + if (!stateFingerprint) { + return "The current segment has no fingerprint metadata. Re-run init_experiment before continuing."; + } + + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + const errors = [...contractResult.errors, ...scriptSnapshot.errors]; + if (errors.length > 0) { + return `${errors.join(" ")} Re-run init_experiment after fixing the workspace contract.`; + } + + const currentFingerprint = buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); + if (currentFingerprint === stateFingerprint) { + return null; + } + + return "autoresearch.md, autoresearch.sh, or autoresearch.checks.sh changed since the current segment was initialized. Re-run init_experiment before continuing."; +} + +export function loadAutoresearchScriptSnapshot(workDir: string): AutoresearchScriptSnapshot { + const benchmarkScriptPath = path.join(workDir, "autoresearch.sh"); + const checksScriptPath = path.join(workDir, "autoresearch.checks.sh"); + const errors: string[] = []; + + let benchmarkScript = ""; + try { + benchmarkScript = fs.readFileSync(benchmarkScriptPath, "utf8"); + } catch { + errors.push(`${benchmarkScriptPath} does not exist. Create it before initializing autoresearch.`); + } + + let checksScript: string | null = null; + try { + checksScript = fs.readFileSync(checksScriptPath, "utf8"); + } catch { + checksScript = null; + } + + return { + benchmarkScript, + benchmarkScriptPath, + checksScript, + checksScriptPath, + errors, + }; +} + +export function normalizeAutoresearchList(values: readonly string[]): string[] { + const normalized: string[] = []; + const seen = new Set(); + for (const value of values) { + const trimmed = value.trim(); + if (trimmed.length === 0) continue; + if (seen.has(trimmed)) continue; + seen.add(trimmed); + normalized.push(trimmed); + } + return normalized; +} + +export function normalizeContractPathSpec(value: string): string { + const normalized = value.trim().replaceAll("\\", "/"); + if (normalized === "." || normalized === "./") return "."; + return normalized.replace(/^\.\/+/, "").replace(/\/+$/, ""); +} + +export function pathMatchesContractPath(pathValue: string, specValue: string): boolean { + const normalizedPath = normalizeContractPathSpec(pathValue); + const normalizedSpec = normalizeContractPathSpec(specValue); + if (normalizedSpec === ".") return true; + return normalizedPath === normalizedSpec || normalizedPath.startsWith(`${normalizedSpec}/`); +} + +export function contractListsEqual(left: readonly string[], right: readonly string[]): boolean { + const normalizedLeft = normalizeAutoresearchList(left); + const normalizedRight = normalizeAutoresearchList(right); + if (normalizedLeft.length !== normalizedRight.length) return false; + return normalizedLeft.every((value, index) => value === normalizedRight[index]); +} + +export function contractPathListsEqual(left: readonly string[], right: readonly string[]): boolean { + const normalizedLeft = normalizeContractPathList(left); + const normalizedRight = normalizeContractPathList(right); + if (normalizedLeft.length !== normalizedRight.length) return false; + return normalizedLeft.every((value, index) => value === normalizedRight[index]); +} + +function createEmptyAutoresearchContract(): AutoresearchContract { + return { + benchmark: { + command: null, + primaryMetric: null, + metricUnit: "", + direction: null, + secondaryMetrics: [], + }, + scopePaths: [], + offLimits: [], + constraints: [], + }; +} + +function normalizeContractPathList(values: readonly string[]): string[] { + return normalizeAutoresearchList(values.map(normalizeContractPathSpec)).sort((left, right) => + left.localeCompare(right), + ); +} + +function extractSections(markdown: string): Map { + const sections = new Map(); + const lines = markdown.split("\n"); + let currentHeading: string | null = null; + let currentLines: string[] = []; + + for (const line of lines) { + const headingMatch = line.match(HEADING_REGEX); + if (headingMatch) { + if (currentHeading) { + sections.set(currentHeading, currentLines.join("\n").trim()); + } + currentHeading = headingMatch[1]?.trim().toLowerCase() ?? null; + currentLines = []; + continue; + } + if (currentHeading) { + currentLines.push(line); + } + } + + if (currentHeading) { + sections.set(currentHeading, currentLines.join("\n").trim()); + } + return sections; +} + +function parseBenchmarkSection(section: string): AutoresearchBenchmarkContract { + const entries = new Map(); + const lines = section.split("\n"); + for (let index = 0; index < lines.length; index += 1) { + const rawLine = lines[index] ?? ""; + const match = rawLine.match(KEY_VALUE_REGEX); + if (!match) continue; + const key = normalizeKey(match[1] ?? ""); + let value = (match[2] ?? "").trim(); + if (key === "secondarymetrics") { + const nestedItems: string[] = []; + for (let nestedIndex = index + 1; nestedIndex < lines.length; nestedIndex += 1) { + const nestedLine = lines[nestedIndex] ?? ""; + if (nestedLine.match(KEY_VALUE_REGEX)) break; + const nestedMatch = nestedLine.match(/^\s{2,}[-*]\s+(.*)$/); + if (!nestedMatch) { + if (nestedLine.trim().length > 0) break; + continue; + } + nestedItems.push((nestedMatch[1] ?? "").trim()); + index = nestedIndex; + } + if (nestedItems.length > 0) { + value = [value, ...nestedItems].filter(Boolean).join(", "); + } + } + entries.set(key, value); + } + + const direction = parseDirection(entries.get("direction")); + return { + command: readNullableEntry(entries.get("command")), + primaryMetric: readNullableEntry(entries.get("primarymetric")), + metricUnit: entries.get("metricunit")?.trim() ?? "", + direction, + secondaryMetrics: parseSecondaryMetrics(entries.get("secondarymetrics")), + }; +} + +function parseListSection(section: string, normalizeItem?: (value: string) => string): string[] { + const items: string[] = []; + let activeItem: string | null = null; + for (const rawLine of section.split("\n")) { + const line = rawLine.trimEnd(); + if (line.trim().length === 0) continue; + const match = rawLine.match(LIST_ITEM_REGEX); + if (match) { + if (activeItem) items.push(activeItem); + activeItem = (match[1] ?? "").trim(); + continue; + } + if (activeItem && /^\s{2,}\S/.test(rawLine)) { + activeItem = `${activeItem} ${line.trim()}`; + continue; + } + if (activeItem) { + items.push(activeItem); + activeItem = null; + } + items.push(line.trim()); + } + if (activeItem) { + items.push(activeItem); + } + const normalizedItems = normalizeAutoresearchList(items); + return normalizeItem ? normalizedItems.map(normalizeItem) : normalizedItems; +} + +function normalizeKey(value: string): string { + return value.toLowerCase().replace(/[^a-z0-9]+/g, ""); +} + +function parseDirection(value: string | undefined): MetricDirection | null { + if (value === "lower" || value === "higher") return value; + return null; +} + +function readNullableEntry(value: string | undefined): string | null { + const trimmed = value?.trim() ?? ""; + return trimmed.length > 0 ? trimmed : null; +} + +function parseSecondaryMetrics(value: string | undefined): string[] { + if (!value) return []; + return normalizeAutoresearchList( + value + .split(",") + .map(entry => entry.trim()) + .filter(Boolean), + ); +} diff --git a/packages/coding-agent/src/autoresearch/dashboard.ts b/packages/coding-agent/src/autoresearch/dashboard.ts index 246277baf..2f46cb268 100644 --- a/packages/coding-agent/src/autoresearch/dashboard.ts +++ b/packages/coding-agent/src/autoresearch/dashboard.ts @@ -1,5 +1,6 @@ import { matchesKey, Text, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui"; import type { Theme } from "../modes/theme/theme"; +import { replaceTabs } from "../tools/render-utils"; import { formatElapsed, formatNum, isBetter } from "./helpers"; import { currentResults, findBaselineMetric, findBaselineRunNumber, findBaselineSecondary } from "./state"; import type { AutoresearchRuntime, DashboardController, ExperimentResult, ExperimentState } from "./types"; @@ -32,7 +33,7 @@ export function createDashboardController(): DashboardController { updateWidget(ctx, runtime): void { if (!ctx.hasUI) return; const state = runtime.state; - if (state.results.length === 0 && !runtime.runningExperiment) { + if (!shouldShowDashboard(runtime, state)) { ctx.ui.setWidget("autoresearch", undefined); return; } @@ -44,8 +45,8 @@ export function createDashboardController(): DashboardController { if (runtime.dashboardExpanded) { const width = process.stdout.columns ?? 120; const lines = [ - renderExpandedHeader(state, width, theme), - ...renderDashboardLines(state, width, theme, 8), + renderExpandedHeader(runtime, width, theme), + ...renderDashboardLines(runtime, width, theme, 8), ]; return new Text(lines.join("\n"), 0, 0); } @@ -53,7 +54,7 @@ export function createDashboardController(): DashboardController { }); }, async showOverlay(ctx, runtime): Promise { - if (!ctx.hasUI || runtime.state.results.length === 0) return; + if (!ctx.hasUI || !shouldShowDashboard(runtime, runtime.state)) return; await ctx.ui.custom( (tui, theme, _keybindings, done) => { overlayTui = tui; @@ -68,8 +69,8 @@ export function createDashboardController(): DashboardController { return { render(width: number): string[] { const terminalRows = process.stdout.rows ?? 40; - const header = renderExpandedHeader(runtime.state, width, theme); - const body = renderDashboardLines(runtime.state, width, theme, 0); + const header = renderExpandedHeader(runtime, width, theme); + const body = renderDashboardLines(runtime, width, theme, 0); if (runtime.runningExperiment) { body.push(renderOverlayRunningLine(runtime, theme, width, spinnerFrame)); } @@ -87,7 +88,7 @@ export function createDashboardController(): DashboardController { }, handleInput(data: string): void { const totalRows = - renderDashboardLines(runtime.state, process.stdout.columns ?? 120, theme, 0).length + + renderDashboardLines(runtime, process.stdout.columns ?? 120, theme, 0).length + (runtime.runningExperiment ? 1 : 0); const viewportRows = Math.max(4, (process.stdout.rows ?? 40) - 4); const maxScroll = Math.max(0, totalRows - viewportRows); @@ -125,41 +126,87 @@ export function createDashboardController(): DashboardController { function renderRunningOnly(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { const parts = [theme.fg("accent", "autoresearch"), theme.fg("warning", " running...")]; if (state.name) { - parts.push(theme.fg("dim", ` | ${state.name}`)); + parts.push(theme.fg("dim", ` | ${replaceTabs(state.name)}`)); } if (runtime.runningExperiment) { - parts.push(theme.fg("dim", ` | ${runtime.runningExperiment.command}`)); + parts.push(theme.fg("dim", ` | ${replaceTabs(runtime.runningExperiment.command)}`)); } return parts.join(""); } -function renderExpandedHeader(state: ExperimentState, width: number, theme: Theme): string { - const label = state.name ? ` autoresearch: ${state.name} ` : " autoresearch "; - const hint = theme.fg("dim", " ctrl+x collapse ctrl+shift+x fullscreen "); +function shouldShowDashboard(runtime: AutoresearchRuntime, state: ExperimentState): boolean { + return ( + runtime.autoresearchMode || + state.results.length > 0 || + runtime.runningExperiment !== null || + runtime.lastRunSummary !== null + ); +} + +function renderExpandedHeader(runtime: AutoresearchRuntime, width: number, theme: Theme): string { + const state = runtime.state; + const status = renderModeStatus(runtime, state); + const label = state.name ? ` autoresearch: ${replaceTabs(state.name)} ` : " autoresearch "; + const hint = theme.fg("dim", ` ctrl+x collapse ctrl+shift+x overlay${status ? ` ${status}` : ""} `); const fillWidth = Math.max(0, width - visibleWidth(label) - visibleWidth(hint)); return truncateToWidth(theme.fg("accent", label) + theme.fg("borderMuted", "-".repeat(fillWidth)) + hint, width); } function renderCollapsedLine(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { + if (runtime.lastRunSummary) { + const parts = [ + theme.fg("accent", "autoresearch"), + theme.fg("warning", ` pending run #${runtime.lastRunSummary.runNumber}`), + theme.fg("dim", runtime.lastRunSummary.passed ? " pass" : " fail"), + ]; + if (runtime.lastRunSummary.parsedPrimary !== null) { + parts.push( + theme.fg( + "muted", + ` | ${state.metricName}=${formatNum(runtime.lastRunSummary.parsedPrimary, state.metricUnit)}`, + ), + ); + } + parts.push(theme.fg("warning", " | log_experiment required")); + if (!runtime.autoresearchMode) { + parts.push(theme.fg("dim", " | mode off")); + } + return parts.join(""); + } + if (state.results.length === 0) { + const modeStatus = runtime.autoresearchMode ? "baseline pending" : "mode off"; + const parts = [theme.fg("accent", "autoresearch"), theme.fg("warning", ` ${modeStatus}`)]; + if (state.name) { + parts.push(theme.fg("dim", ` | ${replaceTabs(state.name)}`)); + } + if (runtime.autoresearchMode) { + parts.push(theme.fg("dim", " | run the baseline")); + } + return parts.join(""); + } const current = currentResults(state.results, state.currentSegment); const kept = current.filter(result => result.status === "keep").length; const crashed = current.filter(result => result.status === "crash").length; const checksFailed = current.filter(result => result.status === "checks_failed").length; const best = findBestResult(state); + const archivedRuns = Math.max(0, state.results.length - current.length); const parts = [ theme.fg("accent", "autoresearch"), - theme.fg("muted", ` ${state.results.length} runs`), + theme.fg("muted", ` ${current.length} runs`), theme.fg("success", ` ${kept} kept`), ]; + if (archivedRuns > 0) parts.push(theme.fg("dim", ` +${archivedRuns} archived`)); if (crashed > 0) parts.push(theme.fg("error", ` ${crashed} crash`)); if (checksFailed > 0) parts.push(theme.fg("error", ` ${checksFailed} checks_failed`)); parts.push(theme.fg("dim", " | ")); - parts.push( - theme.fg( - "warning", - `${state.metricName}: ${formatNum(best?.result.metric ?? state.bestMetric, state.metricUnit)}`, - ), - ); + if (best && state.bestMetric !== null && best.result.metric !== state.bestMetric) { + parts.push(theme.fg("warning", `best ${formatNum(best.result.metric, state.metricUnit)}`)); + parts.push(theme.fg("dim", ` baseline ${formatNum(state.bestMetric, state.metricUnit)}`)); + } else if (state.bestMetric !== null) { + parts.push(theme.fg("warning", `baseline ${formatNum(state.bestMetric, state.metricUnit)}`)); + } else { + parts.push(theme.fg("warning", `no kept runs yet`)); + } if (state.confidence !== null) { const confidenceColor = state.confidence >= 2 ? "success" : state.confidence >= 1 ? "warning" : "error"; parts.push(theme.fg("dim", " | ")); @@ -167,13 +214,42 @@ function renderCollapsedLine(runtime: AutoresearchRuntime, state: ExperimentStat } if (runtime.runningExperiment) { parts.push(theme.fg("dim", ` | running ${formatElapsed(Date.now() - runtime.runningExperiment.startedAt)}`)); + } else if (!runtime.autoresearchMode) { + parts.push(theme.fg("dim", ` | ${renderModeStatus(runtime, state)}`)); } parts.push(theme.fg("dim", " | ctrl+x expand")); return parts.join(""); } -export function renderDashboardLines(state: ExperimentState, width: number, theme: Theme, maxRows: number): string[] { +export function renderDashboardLines( + runtime: AutoresearchRuntime, + width: number, + theme: Theme, + maxRows: number, +): string[] { + const state = runtime.state; if (state.results.length === 0) { + if (runtime.lastRunSummary) { + const lines = [ + truncateToWidth(`Pending run: #${runtime.lastRunSummary.runNumber}`, width), + truncateToWidth( + `Result: ${runtime.lastRunSummary.passed ? "passed" : "failed"}${runtime.lastRunSummary.parsedPrimary !== null ? ` ${state.metricName} ${formatNum(runtime.lastRunSummary.parsedPrimary, state.metricUnit)}` : ""}`, + width, + ), + truncateToWidth("Next action: finish log_experiment before starting another run.", width), + ]; + if (!runtime.autoresearchMode) { + lines.push(truncateToWidth("Mode: off", width)); + } + return lines; + } + if (runtime.autoresearchMode) { + return [ + truncateToWidth("Current segment: 0 runs", width), + truncateToWidth("Baseline: pending", width), + truncateToWidth("Next action: run and log the baseline experiment.", width), + ]; + } return [theme.fg("dim", "No experiments logged yet.")]; } @@ -188,7 +264,7 @@ export function renderDashboardLines(state: ExperimentState, width: number, them const best = findBestResult(state); const lines = [ truncateToWidth( - `Runs: ${state.results.length} ${kept} kept ${discarded} discarded ${crashed} crashed ${checksFailed} checks_failed`, + `Current segment: ${current.length} runs ${kept} kept ${discarded} discarded ${crashed} crashed ${checksFailed} checks_failed`, width, ), truncateToWidth( @@ -196,8 +272,25 @@ export function renderDashboardLines(state: ExperimentState, width: number, them width, ), ]; + if (state.results.length > current.length) { + lines.push( + truncateToWidth(`Archived from earlier segments: ${state.results.length - current.length} runs`, width), + ); + } + if (runtime.lastRunSummary) { + lines.push( + truncateToWidth( + `Pending run: #${runtime.lastRunSummary.runNumber} (${runtime.lastRunSummary.passed ? "passed" : "failed"}) — log_experiment required`, + width, + ), + ); + } + if (!runtime.autoresearchMode) { + lines.push(truncateToWidth(`Mode: ${renderModeStatus(runtime, state)}`, width)); + } if (best) { - let progress = `Best: ${formatNum(best.result.metric, state.metricUnit)} (#${best.index + 1})`; + const bestRunNumber = best.result.runNumber ?? best.index + 1; + let progress = `Best: ${formatNum(best.result.metric, state.metricUnit)} (#${bestRunNumber})`; if (baseline !== null && baseline !== 0 && best.result.metric !== baseline) { const delta = ((best.result.metric - baseline) / baseline) * 100; const sign = delta > 0 ? "+" : ""; @@ -227,9 +320,9 @@ export function renderDashboardLines(state: ExperimentState, width: number, them lines.push(renderTableHeader(state, width, theme)); lines.push(theme.fg("borderMuted", "-".repeat(Math.max(0, width - 1)))); - const visible = maxRows > 0 ? state.results.slice(-maxRows) : state.results; - if (visible.length < state.results.length) { - lines.push(theme.fg("dim", `... ${state.results.length - visible.length} earlier runs hidden ...`)); + const visible = maxRows > 0 ? current.slice(-maxRows) : current; + if (visible.length < current.length) { + lines.push(theme.fg("dim", `... ${current.length - visible.length} earlier runs hidden ...`)); } for (const result of visible) { lines.push(renderResultRow(result, state, baselineSecondary, width, theme)); @@ -252,7 +345,7 @@ function renderResultRow( width: number, theme: Theme, ): string { - const runNumber = state.results.indexOf(result) + 1; + const runNumber = result.runNumber ?? state.results.indexOf(result) + 1; const secondary = state.secondaryMetrics .map(metric => truncateToWidth( @@ -268,7 +361,7 @@ function renderResultRow( `${theme.fg(statusColor, formatNum(result.metric, state.metricUnit).padEnd(12))}` + `${secondary}` + `${theme.fg(statusColor, result.status.padEnd(14))}` + - `${theme.fg("muted", result.description)}`; + `${theme.fg("muted", replaceTabs(result.description))}`; return truncateToWidth(line, width); } @@ -306,7 +399,9 @@ function renderOverlayRunningLine( return truncateToWidth( theme.fg( "warning", - `${spinner} running ${formatElapsed(Date.now() - (runtime.runningExperiment?.startedAt ?? Date.now()))} ${runtime.runningExperiment?.command ?? ""}`, + `${spinner} running ${formatElapsed(Date.now() - (runtime.runningExperiment?.startedAt ?? Date.now()))} ${replaceTabs( + runtime.runningExperiment?.command ?? "", + )}`, ), width, ); @@ -328,6 +423,17 @@ function renderOverlayFooter( return theme.fg("borderMuted", "-".repeat(fill)) + hint; } +function renderModeStatus(runtime: AutoresearchRuntime, state: ExperimentState): string { + if (runtime.autoresearchMode) { + return state.results.length === 0 ? "baseline pending" : "mode on"; + } + const current = currentResults(state.results, state.currentSegment); + if (state.maxExperiments !== null && current.length >= state.maxExperiments) { + return "segment complete"; + } + return "mode off"; +} + function findBestResult(state: ExperimentState): { index: number; result: ExperimentResult } | null { let best: { index: number; result: ExperimentResult } | null = null; for (let index = 0; index < state.results.length; index += 1) { diff --git a/packages/coding-agent/src/autoresearch/git.ts b/packages/coding-agent/src/autoresearch/git.ts index 12caf3721..e22ea4976 100644 --- a/packages/coding-agent/src/autoresearch/git.ts +++ b/packages/coding-agent/src/autoresearch/git.ts @@ -1,5 +1,5 @@ import type { ExtensionAPI } from "../extensibility/extensions"; -import { PROTECTED_AUTORESEARCH_FILES } from "./helpers"; +import { isAutoresearchLocalStatePath, normalizeAutoresearchPath } from "./helpers"; const AUTORESEARCH_BRANCH_PREFIX = "autoresearch/"; const BRANCH_NAME_MAX_LENGTH = 48; @@ -17,6 +17,12 @@ export interface EnsureAutoresearchBranchSuccess { export type EnsureAutoresearchBranchResult = EnsureAutoresearchBranchFailure | EnsureAutoresearchBranchSuccess; +export async function getCurrentAutoresearchBranch(api: ExtensionAPI, workDir: string): Promise { + const currentBranchResult = await api.exec("git", ["branch", "--show-current"], { cwd: workDir, timeout: 5_000 }); + const currentBranch = currentBranchResult.stdout.trim(); + return currentBranch.startsWith(AUTORESEARCH_BRANCH_PREFIX) ? currentBranch : null; +} + export async function ensureAutoresearchBranch( api: ExtensionAPI, workDir: string, @@ -29,19 +35,10 @@ export async function ensureAutoresearchBranch( ok: false, }; } + const repoRoot = repoRootResult.stdout.trim() || workDir; - const currentBranchResult = await api.exec("git", ["branch", "--show-current"], { cwd: workDir, timeout: 5_000 }); - const currentBranch = currentBranchResult.stdout.trim(); - if (currentBranch.startsWith(AUTORESEARCH_BRANCH_PREFIX)) { - return { - branchName: currentBranch, - created: false, - ok: true, - }; - } - - const dirtyPathsResult = await api.exec("git", ["status", "--porcelain", "--untracked-files=all"], { - cwd: workDir, + const dirtyPathsResult = await api.exec("git", ["status", "--porcelain=v1", "-z", "--untracked-files=all"], { + cwd: repoRoot, timeout: 5_000, }); if (dirtyPathsResult.code !== 0) { @@ -51,17 +48,22 @@ export async function ensureAutoresearchBranch( }; } - const unsafeDirtyPaths = parseUnsafeDirtyPaths(dirtyPathsResult.stdout); - if (unsafeDirtyPaths.length > 0) { - const preview = unsafeDirtyPaths.slice(0, 5).join(", "); - const suffix = unsafeDirtyPaths.length > 5 ? ` (+${unsafeDirtyPaths.length - 5} more)` : ""; + const workDirPrefix = await readGitWorkDirPrefix(api, workDir); + const unsafeDirtyPaths = collectUnsafeDirtyPaths(dirtyPathsResult.stdout, workDirPrefix); + const currentBranch = await getCurrentAutoresearchBranch(api, workDir); + if (currentBranch) { + if (unsafeDirtyPaths.length > 0) { + return buildUnsafeDirtyPathsFailure(unsafeDirtyPaths); + } return { - error: - "Autoresearch needs a clean git worktree before it can create an isolated branch. " + - `Commit or stash these paths first: ${preview}${suffix}`, - ok: false, + branchName: currentBranch, + created: false, + ok: true, }; } + if (unsafeDirtyPaths.length > 0) { + return buildUnsafeDirtyPathsFailure(unsafeDirtyPaths); + } const branchName = await allocateBranchName(api, workDir, goal); const checkoutResult = await api.exec("git", ["checkout", "-b", branchName], { cwd: workDir, timeout: 10_000 }); @@ -81,7 +83,69 @@ export async function ensureAutoresearchBranch( }; } -function parseUnsafeDirtyPaths(statusOutput: string): string[] { +export function parseWorkDirDirtyPaths(statusOutput: string, workDirPrefix: string): string[] { + const relativePaths: string[] = []; + for (const dirtyPath of parseDirtyPaths(statusOutput)) { + const relativePath = relativizeGitPathToWorkDir(dirtyPath, workDirPrefix); + if (relativePath === null) continue; + relativePaths.push(relativePath); + } + return relativePaths; +} + +export function relativizeGitPathToWorkDir(repoRelativePath: string, workDirPrefix: string): string | null { + const normalizedPath = normalizeStatusPath(repoRelativePath); + const normalizedPrefix = normalizeAutoresearchPath(workDirPrefix); + if (normalizedPrefix === "" || normalizedPrefix === ".") { + return normalizedPath; + } + if (normalizedPath === normalizedPrefix) { + return "."; + } + if (!normalizedPath.startsWith(`${normalizedPrefix}/`)) { + return null; + } + return normalizeAutoresearchPath(normalizedPath.slice(normalizedPrefix.length + 1)); +} + +async function readGitWorkDirPrefix(api: ExtensionAPI, workDir: string): Promise { + const prefixResult = await api.exec("git", ["rev-parse", "--show-prefix"], { cwd: workDir, timeout: 5_000 }); + if (prefixResult.code !== 0) { + return ""; + } + return prefixResult.stdout.trim(); +} + +export function parseDirtyPaths(statusOutput: string): string[] { + if (statusOutput.includes("\0")) { + return parseDirtyPathsNul(statusOutput); + } + return parseDirtyPathsLines(statusOutput); +} + +function parseDirtyPathsNul(statusOutput: string): string[] { + const unsafePaths = new Set(); + let index = 0; + while (index + 3 <= statusOutput.length) { + const statusToken = statusOutput.slice(index, index + 3); + index += 3; + const pathEnd = statusOutput.indexOf("\0", index); + if (pathEnd < 0) break; + const firstPath = statusOutput.slice(index, pathEnd); + index = pathEnd + 1; + addDirtyPath(unsafePaths, firstPath); + if (isRenameOrCopy(statusToken)) { + const secondPathEnd = statusOutput.indexOf("\0", index); + if (secondPathEnd < 0) break; + const secondPath = statusOutput.slice(index, secondPathEnd); + index = secondPathEnd + 1; + addDirtyPath(unsafePaths, secondPath); + } + } + return [...unsafePaths]; +} + +function parseDirtyPathsLines(statusOutput: string): string[] { const unsafePaths = new Set(); for (const line of statusOutput.split("\n")) { const trimmedLine = line.trimEnd(); @@ -89,23 +153,19 @@ function parseUnsafeDirtyPaths(statusOutput: string): string[] { const rawPath = trimmedLine.slice(3).trim(); if (rawPath.length === 0) continue; const renameParts = rawPath.split(" -> "); - const normalizedPath = normalizeStatusPath(renameParts[renameParts.length - 1] ?? rawPath); - if (normalizedPath.length === 0) continue; - if (PROTECTED_AUTORESEARCH_FILES.some(path => path === normalizedPath)) continue; - unsafePaths.add(normalizedPath); + for (const renamePart of renameParts) { + addDirtyPath(unsafePaths, renamePart); + } } return [...unsafePaths]; } -function normalizeStatusPath(path: string): string { +export function normalizeStatusPath(path: string): string { let normalized = path.trim(); if (normalized.startsWith('"') && normalized.endsWith('"')) { normalized = normalized.slice(1, -1); } - if (normalized.startsWith("./")) { - normalized = normalized.slice(2); - } - return normalized; + return normalizeAutoresearchPath(normalized); } async function allocateBranchName(api: ExtensionAPI, workDir: string, goal: string | null): Promise { @@ -147,3 +207,37 @@ function currentDateStamp(): string { function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { return `${result.stdout}${result.stderr}`; } + +function addDirtyPath(paths: Set, rawPath: string): void { + const normalizedPath = normalizeStatusPath(rawPath); + if (normalizedPath.length === 0) return; + paths.add(normalizedPath); +} + +function buildUnsafeDirtyPathsFailure(unsafeDirtyPaths: string[]): EnsureAutoresearchBranchFailure { + const preview = unsafeDirtyPaths.slice(0, 5).join(", "); + const suffix = unsafeDirtyPaths.length > 5 ? ` (+${unsafeDirtyPaths.length - 5} more)` : ""; + return { + error: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. " + + `Commit or stash these paths first: ${preview}${suffix}`, + ok: false, + }; +} + +function isRenameOrCopy(statusToken: string): boolean { + const trimmed = statusToken.trim(); + return trimmed.startsWith("R") || trimmed.startsWith("C"); +} + +function collectUnsafeDirtyPaths(statusOutput: string, workDirPrefix: string): string[] { + const unsafeDirtyPaths: string[] = []; + for (const dirtyPath of parseDirtyPaths(statusOutput)) { + const relativePath = relativizeGitPathToWorkDir(dirtyPath, workDirPrefix); + if (relativePath && isAutoresearchLocalStatePath(relativePath)) { + continue; + } + unsafeDirtyPaths.push(relativePath ?? normalizeStatusPath(dirtyPath)); + } + return unsafeDirtyPaths; +} diff --git a/packages/coding-agent/src/autoresearch/helpers.ts b/packages/coding-agent/src/autoresearch/helpers.ts index 681c50f41..20f2ae0f7 100644 --- a/packages/coding-agent/src/autoresearch/helpers.ts +++ b/packages/coding-agent/src/autoresearch/helpers.ts @@ -1,21 +1,28 @@ -import * as crypto from "node:crypto"; import * as fs from "node:fs"; -import * as os from "node:os"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; -import type { ASIData, ASIValue, AutoresearchConfig, MetricDirection } from "./types"; +import type { + ASIData, + ASIValue, + AutoresearchConfig, + MetricDirection, + NumericMetricMap, + PendingRunSummary, +} from "./types"; export const METRIC_LINE_PREFIX = "METRIC"; export const ASI_LINE_PREFIX = "ASI"; export const EXPERIMENT_MAX_LINES = 10; export const EXPERIMENT_MAX_BYTES = 4 * 1024; -export const PROTECTED_AUTORESEARCH_FILES = [ - "autoresearch.jsonl", +export const AUTORESEARCH_COMMITTABLE_FILES = [ "autoresearch.md", - "autoresearch.ideas.md", + "autoresearch.program.md", "autoresearch.sh", "autoresearch.checks.sh", + "autoresearch.ideas.md", ] as const; +export const AUTORESEARCH_LOCAL_STATE_FILES = ["autoresearch.jsonl"] as const; +export const AUTORESEARCH_LOCAL_STATE_DIRECTORIES = [".autoresearch"] as const; const DENIED_KEY_NAMES = new Set(["__proto__", "constructor", "prototype"]); @@ -112,21 +119,57 @@ export function formatElapsed(milliseconds: number): string { return `${seconds}s`; } -export function createTempFileAllocator(): () => string { - let tempPath: string | undefined; - return () => { - if (tempPath) return tempPath; - tempPath = path.join(os.tmpdir(), `pi-autoresearch-${crypto.randomUUID()}.log`); - return tempPath; - }; +export function getAutoresearchRunDirectory(workDir: string, runNumber: number): string { + return path.join(workDir, ".autoresearch", "runs", String(runNumber).padStart(4, "0")); } -export function killTree(pid: number): void { +export function getNextAutoresearchRunNumber(workDir: string, lastRunNumber: number | null): number { + const runsDirectory = path.join(workDir, ".autoresearch", "runs"); + let maxRunNumber = lastRunNumber ?? 0; try { - process.kill(-pid, "SIGTERM"); + for (const entry of fs.readdirSync(runsDirectory, { withFileTypes: true })) { + if (!entry.isDirectory()) continue; + const runNumber = Number.parseInt(entry.name, 10); + if (Number.isFinite(runNumber)) { + maxRunNumber = Math.max(maxRunNumber, runNumber); + } + } + } catch (error) { + if (!isEnoent(error)) { + throw error; + } + } + return maxRunNumber + 1; +} + +export function normalizeAutoresearchPath(relativePath: string): string { + const normalized = relativePath.replaceAll("\\", "/").trim(); + if (normalized === "." || normalized === "./") return "."; + return normalized.replace(/^\.\/+/, "").replace(/\/+$/, ""); +} + +export function isAutoresearchCommittableFile(relativePath: string): boolean { + const normalized = normalizeAutoresearchPath(relativePath); + return AUTORESEARCH_COMMITTABLE_FILES.some(candidate => candidate === normalized); +} + +export function isAutoresearchLocalStatePath(relativePath: string): boolean { + const normalized = normalizeAutoresearchPath(relativePath); + if (AUTORESEARCH_LOCAL_STATE_FILES.some(candidate => candidate === normalized)) { + return true; + } + return AUTORESEARCH_LOCAL_STATE_DIRECTORIES.some(candidate => { + const normalizedCandidate = normalizeAutoresearchPath(candidate); + return normalized === normalizedCandidate || normalized.startsWith(`${normalizedCandidate}/`); + }); +} + +export function killTree(pid: number, signal: NodeJS.Signals | number = "SIGTERM"): void { + try { + process.kill(-pid, signal); } catch { try { - process.kill(pid, "SIGTERM"); + process.kill(pid, signal); } catch { // Process already exited. } @@ -159,6 +202,44 @@ export function inferMetricUnitFromName(name: string): string { return ""; } +export async function readPendingRunSummary( + workDir: string, + loggedRunNumbers: ReadonlySet = new Set(), +): Promise { + const runsDir = path.join(workDir, ".autoresearch", "runs"); + let entries: fs.Dirent[]; + try { + entries = await fs.promises.readdir(runsDir, { withFileTypes: true }); + } catch (error) { + if (isEnoent(error)) return null; + throw error; + } + + const runDirectories = entries + .filter(entry => entry.isDirectory()) + .map(entry => entry.name) + .sort((left, right) => right.localeCompare(left)); + + for (const directoryName of runDirectories) { + const runDirectory = path.join(runsDir, directoryName); + const runJsonPath = path.join(runDirectory, "run.json"); + let parsed: unknown; + try { + parsed = await Bun.file(runJsonPath).json(); + } catch (error) { + if (isEnoent(error)) continue; + throw error; + } + + const pendingRun = parsePendingRunSummary(parsed, runDirectory, directoryName, loggedRunNumbers); + if (pendingRun) { + return pendingRun; + } + } + + return null; +} + export function readConfig(cwd: string): AutoresearchConfig { const configPath = path.join(cwd, "autoresearch.config.json"); try { @@ -207,3 +288,139 @@ export function validateWorkDir(cwd: string): string | null { return `workingDir ${workDir} is unavailable.`; } } + +function parsePendingRunSummary( + value: unknown, + runDirectory: string, + directoryName: string, + loggedRunNumbers: ReadonlySet, +): PendingRunSummary | null { + if (typeof value !== "object" || value === null) return null; + const candidate = value as { + checks?: { durationSeconds?: unknown; passed?: unknown; timedOut?: unknown }; + completedAt?: unknown; + command?: unknown; + durationSeconds?: unknown; + exitCode?: unknown; + loggedAt?: unknown; + parsedAsi?: unknown; + parsedMetrics?: unknown; + parsedPrimary?: unknown; + runNumber?: unknown; + status?: unknown; + timedOut?: unknown; + }; + if (candidate.loggedAt !== undefined || candidate.status !== undefined) { + return null; + } + + const command = typeof candidate.command === "string" ? candidate.command : ""; + const runNumber = + typeof candidate.runNumber === "number" && Number.isFinite(candidate.runNumber) + ? candidate.runNumber + : parseInt(directoryName, 10); + if (!Number.isFinite(runNumber)) return null; + if (loggedRunNumbers.has(runNumber)) return null; + + const hasCompletedMetadata = + typeof candidate.completedAt === "string" || + candidate.exitCode !== undefined || + candidate.timedOut !== undefined || + candidate.durationSeconds !== undefined || + candidate.checks !== undefined || + candidate.parsedPrimary !== undefined || + candidate.parsedMetrics !== undefined || + candidate.parsedAsi !== undefined; + if (!hasCompletedMetadata) { + return null; + } + + const checksPass = + typeof candidate.checks?.passed === "boolean" + ? candidate.checks.passed + : typeof candidate.checks?.timedOut === "boolean" && candidate.checks.timedOut + ? false + : null; + const exitCode = + typeof candidate.exitCode === "number" && Number.isFinite(candidate.exitCode) ? candidate.exitCode : null; + const timedOut = candidate.timedOut === true; + const durationSeconds = + typeof candidate.durationSeconds === "number" && Number.isFinite(candidate.durationSeconds) + ? candidate.durationSeconds + : null; + const parsedPrimary = + typeof candidate.parsedPrimary === "number" && Number.isFinite(candidate.parsedPrimary) + ? candidate.parsedPrimary + : null; + const parsedAsi = cloneAsiData(candidate.parsedAsi); + const parsedMetrics = cloneNumericMetricMap(candidate.parsedMetrics); + const checksDurationSeconds = + typeof candidate.checks?.durationSeconds === "number" && Number.isFinite(candidate.checks.durationSeconds) + ? candidate.checks.durationSeconds + : null; + const checksTimedOut = candidate.checks?.timedOut === true; + + return { + checksDurationSeconds, + checksPass, + checksTimedOut, + command, + durationSeconds, + parsedAsi, + parsedMetrics, + parsedPrimary, + passed: exitCode === 0 && !timedOut && checksPass !== false, + runDirectory, + runNumber, + }; +} + +function cloneNumericMetricMap(value: unknown): NumericMetricMap | null { + if (typeof value !== "object" || value === null) return null; + const metrics = value as { [key: string]: unknown }; + const clone: NumericMetricMap = {}; + for (const [key, entryValue] of Object.entries(metrics)) { + if (typeof entryValue === "number" && Number.isFinite(entryValue)) { + clone[key] = entryValue; + } + } + return Object.keys(clone).length > 0 ? clone : null; +} + +function cloneAsiData(value: unknown): ASIData | null { + if (typeof value !== "object" || value === null) return null; + const candidate = value as { [key: string]: unknown }; + const clone: ASIData = {}; + for (const [key, entryValue] of Object.entries(candidate)) { + const sanitized = clonePendingAsiValue(entryValue); + if (sanitized !== undefined) { + clone[key] = sanitized; + } + } + return Object.keys(clone).length > 0 ? clone : null; +} + +function clonePendingAsiValue(value: unknown): ASIValue | undefined { + if (value === null) return null; + if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") { + return value; + } + if (Array.isArray(value)) { + const items = value + .map(entry => clonePendingAsiValue(entry)) + .filter((entry): entry is NonNullable => entry !== undefined); + return items; + } + if (typeof value === "object") { + const candidate = value as { [key: string]: unknown }; + const clone: { [key: string]: ASIValue } = {}; + for (const [key, entryValue] of Object.entries(candidate)) { + const sanitized = clonePendingAsiValue(entryValue); + if (sanitized !== undefined) { + clone[key] = sanitized; + } + } + return clone; + } + return undefined; +} diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 6de676653..4ddcbd2c2 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -5,27 +5,49 @@ import { renderPromptTemplate } from "../config/prompt-templates"; import type { ExtensionContext, ExtensionFactory } from "../extensibility/extensions"; import commandInitializeTemplate from "./command-initialize.md" with { type: "text" }; import commandResumeTemplate from "./command-resume.md" with { type: "text" }; +import { pathMatchesContractPath } from "./contract"; import { createDashboardController } from "./dashboard"; import { ensureAutoresearchBranch } from "./git"; -import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "./helpers"; +import { + formatNum, + isAutoresearchCommittableFile, + isAutoresearchLocalStatePath, + isAutoresearchShCommand, + normalizeAutoresearchPath, + readMaxExperiments, + readPendingRunSummary, + resolveWorkDir, + validateWorkDir, +} from "./helpers"; import promptTemplate from "./prompt.md" with { type: "text" }; import resumeMessageTemplate from "./resume-message.md" with { type: "text" }; import { cloneExperimentState, createExperimentState, createRuntimeStore, + currentResults, + findBaselineMetric, reconstructControlState, reconstructStateFromJsonl, } from "./state"; import { createInitExperimentTool } from "./tools/init-experiment"; import { createLogExperimentTool } from "./tools/log-experiment"; import { createRunExperimentTool } from "./tools/run-experiment"; -import type { AutoresearchRuntime } from "./types"; +import type { AutoresearchRuntime, ChecksResult, ExperimentResult, PendingRunSummary } from "./types"; -const AUTORESUME_INTERVAL_MS = 5 * 60 * 1000; -const MAX_AUTORESUME_TURNS = 20; const EXPERIMENT_TOOL_NAMES = ["init_experiment", "run_experiment", "log_experiment"]; +interface AutoresearchSetupInput { + intent: string; + benchmarkCommand: string; + metricName: string; + metricUnit: string; + direction: "lower" | "higher"; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; +} + export const createAutoresearchExtension: ExtensionFactory = api => { const runtimeStore = createRuntimeStore(); const dashboard = createDashboardController(); @@ -37,17 +59,18 @@ export const createAutoresearchExtension: ExtensionFactory = api => { const runtime = getRuntime(ctx); const workDir = resolveWorkDir(ctx.cwd); const reconstructed = reconstructStateFromJsonl(workDir); - const control = reconstructControlState(ctx.sessionManager.getEntries()); + const control = reconstructControlState(ctx.sessionManager.getBranch()); + const loggedRunNumbers = collectLoggedRunNumbers(reconstructed.state.results); runtime.state = cloneExperimentState(reconstructed.state); runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); runtime.goal = control.goal; runtime.autoresearchMode = control.autoresearchMode; - runtime.lastAutoResumeTime = 0; - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; - runtime.lastRunChecks = null; - runtime.lastRunDuration = null; - runtime.lastRunAsi = null; + runtime.lastRunSummary = await readPendingRunSummary(workDir, loggedRunNumbers); + runtime.lastRunChecks = summaryToChecks(runtime.lastRunSummary); + runtime.lastRunDuration = runtime.lastRunSummary?.durationSeconds ?? null; + runtime.lastRunAsi = runtime.lastRunSummary?.parsedAsi ?? null; + runtime.lastRunArtifactDir = runtime.lastRunSummary?.runDirectory ?? null; + runtime.lastRunNumber = runtime.lastRunSummary?.runNumber ?? null; runtime.runningExperiment = null; dashboard.updateWidget(ctx, runtime); const activeTools = api.getActiveTools(); @@ -78,6 +101,49 @@ export const createAutoresearchExtension: ExtensionFactory = api => { api.registerTool(createInitExperimentTool({ dashboard, getRuntime, pi: api })); api.registerTool(createRunExperimentTool({ dashboard, getRuntime, pi: api })); api.registerTool(createLogExperimentTool({ dashboard, getRuntime, pi: api })); + api.on("tool_call", (event, ctx) => { + const runtime = getRuntime(ctx); + if (!runtime.autoresearchMode) return; + if (event.toolName === "bash") { + const command = typeof event.input.command === "string" ? event.input.command : ""; + const validationError = validateAutoresearchBashCommand(command); + if (validationError) { + return { + block: true, + reason: validationError, + }; + } + return; + } + if (event.toolName !== "write" && event.toolName !== "edit" && event.toolName !== "ast_edit") return; + + const rawPaths = getGuardedToolPaths(event.toolName, event.input); + if (rawPaths === null) { + return { + block: true, + reason: + "Autoresearch requires an explicit target path for this editing tool so it can enforce Files in Scope and Off Limits before changes are made.", + }; + } + + const workDir = resolveWorkDir(ctx.cwd); + for (const rawPath of rawPaths) { + const relativePath = resolveAutoresearchRelativePath(workDir, rawPath); + if (!relativePath.ok) { + return { + block: true, + reason: relativePath.reason, + }; + } + const validationError = validateEditableAutoresearchPath(relativePath.relativePath, runtime); + if (validationError) { + return { + block: true, + reason: `Autoresearch blocked edits to ${relativePath.relativePath}: ${validationError}`, + }; + } + } + }); api.registerCommand("autoresearch", { description: "Start, stop, or clear builtin autoresearch mode.", @@ -102,8 +168,6 @@ export const createAutoresearchExtension: ExtensionFactory = api => { if (trimmed === "off") { setMode(ctx, false, runtime.goal, "off"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; dashboard.updateWidget(ctx, runtime); const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); @@ -113,34 +177,48 @@ export const createAutoresearchExtension: ExtensionFactory = api => { if (trimmed === "clear") { const workDir = resolveWorkDir(ctx.cwd); const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + const localStatePath = path.join(workDir, ".autoresearch"); if (fs.existsSync(jsonlPath)) { fs.rmSync(jsonlPath); } + if (fs.existsSync(localStatePath)) { + fs.rmSync(localStatePath, { force: true, recursive: true }); + } runtime.state = createExperimentState(); runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); runtime.goal = null; + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = null; + runtime.lastRunNumber = null; + runtime.lastRunSummary = null; setMode(ctx, false, null, "clear"); dashboard.updateWidget(ctx, runtime); const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); - ctx.ui.notify("Autoresearch log cleared", "info"); + ctx.ui.notify("Autoresearch local state cleared", "info"); return; } const workDir = resolveWorkDir(ctx.cwd); const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const hasAutoresearchMd = fs.existsSync(autoresearchMdPath); + const controlState = reconstructControlState(ctx.sessionManager.getBranch()); + const shouldResumeExistingNotes = + hasAutoresearchMd && + (hasLocalAutoresearchState(workDir) || (controlState.lastMode !== "clear" && trimmed.length === 0)); - if (hasAutoresearchMd) { - const branchResult = await ensureAutoresearchBranch(api, workDir, runtime.goal); + if (shouldResumeExistingNotes) { + const resumeContext = trimmed; + const resumeGoal = runtime.goal ?? runtime.state.name ?? null; + const branchResult = await ensureAutoresearchBranch(api, workDir, resumeGoal); if (!branchResult.ok) { ctx.ui.notify(branchResult.error, "error"); return; } - setMode(ctx, true, runtime.goal, "on"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; + setMode(ctx, true, resumeGoal, "on"); dashboard.updateWidget(ctx, runtime); await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); api.sendUserMessage( @@ -149,32 +227,34 @@ export const createAutoresearchExtension: ExtensionFactory = api => { branch_status_line: branchResult.created ? `Created and checked out dedicated git branch \`${branchResult.branchName}\` before resuming.` : `Using dedicated git branch \`${branchResult.branchName}\`.`, + has_resume_context: resumeContext.length > 0, + resume_context: resumeContext, }), ); return; } - const intentInput = await ctx.ui.input( - "Autoresearch Intent", + const setup = await promptForAutoresearchSetup( + ctx, trimmed || runtime.goal || "what should autoresearch improve?", ); - if (intentInput === undefined) return; + if (!setup) return; - const intent = intentInput.trim(); - if (intent.length === 0) { - ctx.ui.notify("Autoresearch intent is required", "info"); - return; - } - - const branchResult = await ensureAutoresearchBranch(api, workDir, intent); + const branchResult = await ensureAutoresearchBranch(api, workDir, setup.intent); if (!branchResult.ok) { ctx.ui.notify(branchResult.error, "error"); return; } - setMode(ctx, true, intent, "on"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; + setMode(ctx, true, setup.intent, "on"); + runtime.state.name = setup.intent; + runtime.state.metricName = setup.metricName; + runtime.state.metricUnit = setup.metricUnit; + runtime.state.bestDirection = setup.direction; + runtime.state.benchmarkCommand = setup.benchmarkCommand; + runtime.state.scopePaths = [...setup.scopePaths]; + runtime.state.offLimits = [...setup.offLimits]; + runtime.state.constraints = [...setup.constraints]; dashboard.updateWidget(ctx, runtime); await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); api.sendUserMessage( @@ -182,7 +262,19 @@ export const createAutoresearchExtension: ExtensionFactory = api => { branch_status_line: branchResult.created ? `Created and checked out dedicated git branch \`${branchResult.branchName}\`.` : `Using dedicated git branch \`${branchResult.branchName}\`.`, - intent, + intent: setup.intent, + benchmark_command: setup.benchmarkCommand, + metric_name: setup.metricName, + metric_unit: setup.metricUnit, + direction: setup.direction, + scope_paths: setup.scopePaths, + scope_paths_block: formatBulletBlock(setup.scopePaths, value => ` - \`${value}\``), + has_off_limits: setup.offLimits.length > 0, + off_limits: setup.offLimits, + off_limits_block: formatBulletBlock(setup.offLimits, value => ` - \`${value}\``, " - `(none)`"), + has_constraints: setup.constraints.length > 0, + constraints: setup.constraints, + constraints_block: formatBulletBlock(setup.constraints, value => ` - ${value}`, " - `(none)`"), }), ); }, @@ -217,52 +309,358 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtimeStore.clear(getSessionKey(ctx)); }); - api.on("agent_start", (_event, ctx) => { - getRuntime(ctx).experimentsThisSession = 0; - }); - - api.on("agent_end", (_event, ctx) => { + api.on("agent_end", async (_event, ctx) => { const runtime = getRuntime(ctx); runtime.runningExperiment = null; dashboard.updateWidget(ctx, runtime); dashboard.requestRender(); if (!runtime.autoresearchMode) return; - if (runtime.experimentsThisSession === 0) return; - if (runtime.autoResumeTurns >= MAX_AUTORESUME_TURNS) return; - const now = Date.now(); - if (now - runtime.lastAutoResumeTime < AUTORESUME_INTERVAL_MS) return; - runtime.lastAutoResumeTime = now; - runtime.autoResumeTurns += 1; + if (ctx.hasPendingMessages()) return; const workDir = resolveWorkDir(ctx.cwd); + const pendingRun = + runtime.lastRunSummary ?? + (await readPendingRunSummary(workDir, collectLoggedRunNumbers(runtime.state.results))); + runtime.lastRunSummary = pendingRun; + runtime.lastRunChecks = summaryToChecks(pendingRun); + runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; + runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi; + const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const ideasPath = path.join(workDir, "autoresearch.ideas.md"); - api.sendUserMessage( - renderPromptTemplate(resumeMessageTemplate, { - has_ideas: fs.existsSync(ideasPath), - }), - { deliverAs: "followUp" }, + api.sendMessage( + { + customType: "autoresearch-resume", + content: renderPromptTemplate(resumeMessageTemplate, { + autoresearch_md_path: autoresearchMdPath, + has_ideas: fs.existsSync(ideasPath), + has_pending_run: Boolean(pendingRun), + }), + display: false, + attribution: "agent", + }, + { deliverAs: "nextTurn", triggerTurn: true }, ); }); - api.on("before_agent_start", (event, ctx) => { + api.on("before_agent_start", async (event, ctx) => { const runtime = getRuntime(ctx); if (!runtime.autoresearchMode) return; const workDir = resolveWorkDir(ctx.cwd); const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const checksPath = path.join(workDir, "autoresearch.checks.sh"); const ideasPath = path.join(workDir, "autoresearch.ideas.md"); + const programPath = path.join(workDir, "autoresearch.program.md"); + const pendingRun = + runtime.lastRunSummary ?? + (await readPendingRunSummary(workDir, collectLoggedRunNumbers(runtime.state.results))); + runtime.lastRunSummary = pendingRun; + runtime.lastRunChecks = summaryToChecks(pendingRun); + runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; + runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi; + const currentSegmentResults = currentResults(runtime.state.results, runtime.state.currentSegment); + const baselineMetric = findBaselineMetric(runtime.state.results, runtime.state.currentSegment); + const bestResult = findBestResult(runtime); + const goal = runtime.goal ?? runtime.state.name ?? ""; + const recentResults = currentSegmentResults.slice(-3).map(result => { + const asiSummary = summarizeExperimentAsi(result); + return { + asi_summary: asiSummary, + description: result.description, + has_asi_summary: Boolean(asiSummary), + metric_display: formatNum(result.metric, runtime.state.metricUnit), + run_number: result.runNumber ?? runtime.state.results.indexOf(result) + 1, + status: result.status, + }; + }); return { systemPrompt: renderPromptTemplate(promptTemplate, { base_system_prompt: event.systemPrompt, - goal: runtime.goal ?? event.prompt, + has_goal: goal.trim().length > 0, + goal, working_dir: workDir, default_metric_name: runtime.state.metricName, + metric_name: runtime.state.metricName, has_autoresearch_md: fs.existsSync(autoresearchMdPath), autoresearch_md_path: autoresearchMdPath, has_checks: fs.existsSync(checksPath), checks_path: checksPath, has_ideas: fs.existsSync(ideasPath), ideas_path: ideasPath, + has_program: fs.existsSync(programPath), + program_path: programPath, + current_segment: runtime.state.currentSegment + 1, + current_segment_run_count: currentSegmentResults.length, + has_baseline_metric: baselineMetric !== null, + baseline_metric_display: formatNum(baselineMetric, runtime.state.metricUnit), + has_best_result: Boolean(bestResult), + best_metric_display: bestResult + ? formatNum(bestResult.metric, runtime.state.metricUnit) + : formatNum(baselineMetric, runtime.state.metricUnit), + best_run_number: bestResult + ? (bestResult.runNumber ?? runtime.state.results.indexOf(bestResult) + 1) + : null, + has_recent_results: recentResults.length > 0, + recent_results: recentResults, + has_pending_run: Boolean(pendingRun), + pending_run_number: pendingRun?.runNumber, + pending_run_command: pendingRun?.command, + pending_run_directory: pendingRun?.runDirectory, + pending_run_passed: pendingRun?.passed ?? false, + has_pending_run_metric: pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined, + pending_run_metric_display: + pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined + ? formatNum(pendingRun.parsedPrimary, runtime.state.metricUnit) + : null, }), }; }); }; + +async function promptForAutoresearchSetup( + ctx: ExtensionContext, + defaultIntent: string, +): Promise { + const intentInput = await ctx.ui.input("Autoresearch Intent", defaultIntent); + if (intentInput === undefined) return undefined; + const intent = intentInput.trim(); + if (intent.length === 0) { + ctx.ui.notify("Autoresearch intent is required", "info"); + return undefined; + } + + const benchmarkCommandInput = await ctx.ui.input("Benchmark Command", "bash autoresearch.sh"); + if (benchmarkCommandInput === undefined) return undefined; + const benchmarkCommand = benchmarkCommandInput.trim(); + if (benchmarkCommand.length === 0) { + ctx.ui.notify("Benchmark command is required", "info"); + return undefined; + } + if (!isAutoresearchShCommand(benchmarkCommand)) { + ctx.ui.notify("Benchmark command must invoke `autoresearch.sh` directly", "info"); + return undefined; + } + + const metricNameInput = await ctx.ui.input("Primary Metric Name", "runtime_ms"); + if (metricNameInput === undefined) return undefined; + const metricName = metricNameInput.trim(); + if (metricName.length === 0) { + ctx.ui.notify("Primary metric name is required", "info"); + return undefined; + } + + const metricUnitInput = await ctx.ui.input("Metric Unit", "ms"); + if (metricUnitInput === undefined) return undefined; + const metricUnit = metricUnitInput.trim(); + + const directionInput = await ctx.ui.input("Metric Direction", "lower"); + if (directionInput === undefined) return undefined; + const normalizedDirection = directionInput.trim().toLowerCase(); + if (normalizedDirection !== "lower" && normalizedDirection !== "higher") { + ctx.ui.notify("Metric direction must be `lower` or `higher`", "info"); + return undefined; + } + + const scopePathsInput = await ctx.ui.input("Files in Scope", "packages/coding-agent/src/autoresearch"); + if (scopePathsInput === undefined) return undefined; + const scopePaths = splitSetupList(scopePathsInput); + if (scopePaths.length === 0) { + ctx.ui.notify("Files in Scope must include at least one path", "info"); + return undefined; + } + + const offLimitsInput = await ctx.ui.input("Off Limits", ""); + if (offLimitsInput === undefined) return undefined; + const constraintsInput = await ctx.ui.input("Constraints", ""); + if (constraintsInput === undefined) return undefined; + + return { + intent, + benchmarkCommand, + metricName, + metricUnit, + direction: normalizedDirection, + scopePaths, + offLimits: splitSetupList(offLimitsInput), + constraints: splitSetupList(constraintsInput), + }; +} + +function splitSetupList(value: string): string[] { + return value + .split(/\r?\n|,/) + .map(entry => entry.trim()) + .filter((entry, index, values) => entry.length > 0 && values.indexOf(entry) === index); +} + +function formatBulletBlock(values: string[], renderValue: (value: string) => string, emptyValue = ""): string { + if (values.length === 0) { + return emptyValue; + } + return values.map(renderValue).join("\n"); +} + +function hasLocalAutoresearchState(workDir: string): boolean { + return fs.existsSync(path.join(workDir, "autoresearch.jsonl")) || fs.existsSync(path.join(workDir, ".autoresearch")); +} + +function summarizeExperimentAsi(result: ExperimentResult): string | null { + const hypothesis = typeof result.asi?.hypothesis === "string" ? result.asi.hypothesis.trim() : ""; + const rollbackReason = typeof result.asi?.rollback_reason === "string" ? result.asi.rollback_reason.trim() : ""; + const nextActionHint = typeof result.asi?.next_action_hint === "string" ? result.asi.next_action_hint.trim() : ""; + const summary = [hypothesis, rollbackReason, nextActionHint].filter(part => part.length > 0).join(" | "); + return summary.length > 0 ? summary.slice(0, 220) : null; +} + +function getGuardedToolPaths(toolName: string, input: Record): string[] | null { + if (toolName === "write") { + return typeof input.path === "string" ? [input.path] : null; + } + if (toolName === "ast_edit") { + return typeof input.path === "string" ? [input.path] : null; + } + if (toolName !== "edit") { + return []; + } + + const paths: string[] = []; + if (typeof input.path === "string") { + paths.push(input.path); + } + if (typeof input.rename === "string") { + paths.push(input.rename); + } + if (typeof input.move === "string") { + paths.push(input.move); + } + return paths; +} + +function resolveAutoresearchRelativePath( + workDir: string, + rawPath: string, +): { ok: false; reason: string } | { ok: true; relativePath: string } { + if (looksLikeInternalUrl(rawPath)) { + return { + ok: false, + reason: `Autoresearch cannot validate internal URL paths during scoped editing: ${rawPath}`, + }; + } + const resolvedPath = path.isAbsolute(rawPath) ? path.resolve(rawPath) : path.resolve(workDir, rawPath); + const canonicalWorkDir = canonicalizeExistingPath(workDir); + const canonicalTargetPath = canonicalizeTargetPath(resolvedPath); + const relativePath = path.relative(canonicalWorkDir, canonicalTargetPath); + if (relativePath === ".." || relativePath.startsWith(`..${path.sep}`) || path.isAbsolute(relativePath)) { + return { + ok: false, + reason: `Autoresearch blocked edits outside the working tree: ${rawPath}`, + }; + } + return { + ok: true, + relativePath: relativePath.length === 0 ? "." : normalizeAutoresearchPath(relativePath), + }; +} + +function validateEditableAutoresearchPath(relativePath: string, runtime: AutoresearchRuntime): string | null { + if (isAutoresearchLocalStatePath(relativePath)) { + return "autoresearch local state files are managed by the experiment tools and cannot be edited directly"; + } + if (runtime.state.offLimits.some(spec => pathMatchesContractPath(relativePath, spec))) { + return "this path is listed under Off Limits in autoresearch.md"; + } + if (isAutoresearchCommittableFile(relativePath)) { + return null; + } + if (runtime.state.scopePaths.length === 0) { + return "Files in Scope is not initialized yet; only autoresearch control files may be edited before init_experiment runs"; + } + if (!runtime.state.scopePaths.some(spec => pathMatchesContractPath(relativePath, spec))) { + return "this path is outside Files in Scope in autoresearch.md"; + } + return null; +} + +function findBestResult(runtime: AutoresearchRuntime): ExperimentResult | null { + let best: ExperimentResult | null = null; + for (const result of runtime.state.results) { + if (result.segment !== runtime.state.currentSegment || result.status !== "keep") continue; + if (!best) { + best = result; + continue; + } + if (runtime.state.bestDirection === "lower" ? result.metric < best.metric : result.metric > best.metric) { + best = result; + } + } + return best; +} + +function collectLoggedRunNumbers(results: ExperimentResult[]): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} + +function summaryToChecks(summary: PendingRunSummary | null): ChecksResult | null { + if (!summary || summary.checksPass === null) { + return null; + } + return { + pass: summary.checksPass, + output: "", + duration: summary.checksDurationSeconds ?? 0, + }; +} + +function looksLikeInternalUrl(value: string): boolean { + return /^[a-z][a-z0-9+.-]*:\/\//i.test(value); +} + +function canonicalizeExistingPath(targetPath: string): string { + try { + return fs.realpathSync.native(targetPath); + } catch { + return path.resolve(targetPath); + } +} + +function canonicalizeTargetPath(targetPath: string): string { + const pendingSegments: string[] = []; + let currentPath = path.resolve(targetPath); + while (!fs.existsSync(currentPath)) { + const parentPath = path.dirname(currentPath); + if (parentPath === currentPath) { + return currentPath; + } + pendingSegments.unshift(path.basename(currentPath)); + currentPath = parentPath; + } + return path.resolve(canonicalizeExistingPath(currentPath), ...pendingSegments); +} + +function validateAutoresearchBashCommand(command: string): string | null { + const trimmed = command.trim(); + if (trimmed.length === 0) { + return null; + } + const mutationPatterns = [ + /(^|[;&|()]\s*)(?:bash|sh)\b/, + /(^|[;&|()]\s*)(?:python|python3|node|perl|ruby|php)\b/, + /(^|[;&|()]\s*)(?:mv|cp|rm|mkdir|touch|chmod|chown|ln|install|patch)\b/, + /(^|[;&|()]\s*)sed\s+-i\b/, + /(^|[;&|()]\s*)git\s+(?:add|apply|checkout|clean|commit|merge|rebase|reset|restore|revert|stash|switch|worktree)\b/, + /(^|[^<])>>?/, + /\|\s*tee\b/, + /<< pattern.test(trimmed))) { + return ( + "Autoresearch only allows read-only shell inspection. " + + "Use write/edit/ast_edit for file changes and run_experiment for benchmark execution." + ); + } + return null; +} diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md index 2ed34f189..c02c20f13 100644 --- a/packages/coding-agent/src/autoresearch/prompt.md +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -4,13 +4,60 @@ Autoresearch mode is active. +{{#if has_goal}} Primary goal: {{goal}} +{{else}} +Primary goal is documented in `autoresearch.md` for this session. +{{/if}} Working directory: `{{working_dir}}` You are running an autonomous experiment loop. Keep iterating until the user interrupts you or the configured maximum iteration count is reached. +{{#if has_program}} + +### Local Playbook + +`autoresearch.program.md` exists at `{{program_path}}`. + +Use it as a repo-local strategy overlay for this session. `autoresearch.md` remains the source of truth for benchmark, scope, and constraints. +{{/if}} +{{#if has_recent_results}} + +### Current Segment Snapshot + +- segment: `{{current_segment}}` +- runs in current segment: `{{current_segment_run_count}}` +{{#if has_baseline_metric}} +- baseline `{{metric_name}}`: `{{baseline_metric_display}}` +{{/if}} +{{#if has_best_result}} +- best kept `{{metric_name}}`: `{{best_metric_display}}`{{#if best_run_number}} from run `#{{best_run_number}}`{{/if}} +{{/if}} + +Recent runs: +{{#each recent_results}} +- run `#{{run_number}}`: `{{status}}` `{{metric_display}}` — {{description}} +{{#if has_asi_summary}} + ASI: {{asi_summary}} +{{/if}} +{{/each}} +{{/if}} +{{#if has_pending_run}} + +### Pending Run + +An unlogged run artifact exists at `{{pending_run_directory}}`. + +- run: `#{{pending_run_number}}` +- command: `{{pending_run_command}}` +{{#if has_pending_run_metric}} +- parsed `{{metric_name}}`: `{{pending_run_metric_display}}` +{{/if}} +- result status: {{#if pending_run_passed}}passed{{else}}failed{{/if}} +- finish the `log_experiment` step before starting another benchmark +{{/if}} ### Available tools @@ -80,12 +127,18 @@ Suggested structure: # Autoresearch ## Goal +{{#if has_goal}} - {{goal}} +{{else}} +- document the active target here before the first benchmark +{{/if}} ## Benchmark -- command: -- primary metric: -- secondary metrics: + - command: + - primary metric: + - metric unit: + - direction: + - secondary metrics: memory_mb, rss_mb ## Files in Scope - path: @@ -104,8 +157,9 @@ Suggested structure: - metric: - why it won: -## Ideas -- item +## What's Been Tried +- experiment: +- lesson: ``` ### Guardrails @@ -114,6 +168,7 @@ Suggested structure: - Do not overfit to synthetic inputs if the real workload is broader. - Preserve correctness. - Only modify files that are explicitly in scope for the current session. +- Do not use the general shell tool for file mutations during autoresearch. Use `write`, `edit`, or `ast_edit` for scoped code changes and `run_experiment` for benchmark execution. - If you create `autoresearch.checks.sh`, treat it as a hard gate for `keep`. - If the user sends another message while a run is in progress, finish the current run and logging cycle first, then address the new input in the next iteration. diff --git a/packages/coding-agent/src/autoresearch/resume-message.md b/packages/coding-agent/src/autoresearch/resume-message.md index 62c10b26a..64c4816f3 100644 --- a/packages/coding-agent/src/autoresearch/resume-message.md +++ b/packages/coding-agent/src/autoresearch/resume-message.md @@ -1,8 +1,13 @@ -The autoresearch loop ended unexpectedly. Resume it now. +Continue the autoresearch loop now. + +@{{autoresearch_md_path}} - Read `autoresearch.md` and `autoresearch.jsonl`. - Treat `autoresearch.md` as the source of truth for the current direction, scope, and constraints. - Inspect recent git history for context. +{{#if has_pending_run}} +- Inspect the latest unlogged `run.json` under `.autoresearch/runs/` and finish the pending `log_experiment` step before starting a new benchmark. +{{/if}} - Continue from the most promising unfinished direction. {{#if has_ideas}} - Review `autoresearch.ideas.md` for promising next steps and prune stale items. diff --git a/packages/coding-agent/src/autoresearch/state.ts b/packages/coding-agent/src/autoresearch/state.ts index 9d302d9b9..725a60715 100644 --- a/packages/coding-agent/src/autoresearch/state.ts +++ b/packages/coding-agent/src/autoresearch/state.ts @@ -1,6 +1,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import type { SessionEntry } from "../session/session-manager"; +import { normalizeAutoresearchList, normalizeContractPathSpec } from "./contract"; import { inferMetricUnitFromName, isBetter } from "./helpers"; import type { AutoresearchControlEntryData, @@ -29,6 +30,11 @@ export function createExperimentState(): ExperimentState { currentSegment: 0, maxExperiments: null, confidence: null, + benchmarkCommand: null, + scopePaths: [], + offLimits: [], + constraints: [], + segmentFingerprint: null, }; } @@ -36,12 +42,12 @@ export function createSessionRuntime(): AutoresearchRuntime { return { autoresearchMode: false, dashboardExpanded: false, - lastAutoResumeTime: 0, - experimentsThisSession: 0, - autoResumeTurns: 0, lastRunChecks: null, lastRunDuration: null, lastRunAsi: null, + lastRunArtifactDir: null, + lastRunNumber: null, + lastRunSummary: null, runningExperiment: null, state: createExperimentState(), goal: null, @@ -57,6 +63,9 @@ export function cloneExperimentState(state: ExperimentState): ExperimentState { asi: result.asi ? structuredClone(result.asi) : undefined, })), secondaryMetrics: state.secondaryMetrics.map(metric => ({ ...metric })), + scopePaths: [...state.scopePaths], + offLimits: [...state.offLimits], + constraints: [...state.constraints], }; } @@ -64,13 +73,35 @@ export function currentResults(results: ExperimentResult[], segment: number): Ex return results.filter(result => result.segment === segment); } +export function findBaselineResult(results: ExperimentResult[], segment: number): ExperimentResult | null { + return currentResults(results, segment).find(result => result.status === "keep") ?? null; +} + export function findBaselineMetric(results: ExperimentResult[], segment: number): number | null { - const baseline = results.find(result => result.segment === segment); + const baseline = findBaselineResult(results, segment); return baseline ? baseline.metric : null; } +export function findBestKeptMetric( + results: ExperimentResult[], + segment: number, + direction: MetricDirection, +): number | null { + let best: number | null = null; + for (const result of currentResults(results, segment)) { + if (result.status !== "keep") continue; + if (best === null || isBetter(result.metric, best, direction)) { + best = result.metric; + } + } + return best; +} + export function findBaselineRunNumber(results: ExperimentResult[], segment: number): number | null { - const index = results.findIndex(result => result.segment === segment); + const baseline = findBaselineResult(results, segment); + if (!baseline) return null; + if (baseline.runNumber !== null) return baseline.runNumber; + const index = results.indexOf(baseline); return index >= 0 ? index + 1 : null; } @@ -79,7 +110,7 @@ export function findBaselineSecondary( segment: number, knownMetrics: MetricDef[], ): NumericMetricMap { - const baseline = currentResults(results, segment)[0]; + const baseline = findBaselineResult(results, segment); const values: NumericMetricMap = baseline ? { ...baseline.metrics } : {}; for (const metric of knownMetrics) { if (values[metric.name] !== undefined) continue; @@ -155,22 +186,30 @@ export function reconstructStateFromJsonl(workDir: string): ReconstructedExperim continue; } - if (isConfigEntry(parsed)) { + const configEntry = parseConfigEntry(parsed); + if (configEntry) { if (sawConfig || state.results.length > 0) { segment += 1; } sawConfig = true; state.currentSegment = segment; - if (parsed.name) state.name = parsed.name; - if (parsed.metricName) state.metricName = parsed.metricName; - if (parsed.metricUnit !== undefined) state.metricUnit = parsed.metricUnit; - if (parsed.bestDirection) state.bestDirection = parsed.bestDirection; - state.secondaryMetrics = []; + if (configEntry.name) state.name = configEntry.name; + if (configEntry.metricName) state.metricName = configEntry.metricName; + if (configEntry.metricUnit !== undefined) state.metricUnit = configEntry.metricUnit; + if (configEntry.bestDirection) state.bestDirection = configEntry.bestDirection; + if (configEntry.benchmarkCommand !== undefined) state.benchmarkCommand = configEntry.benchmarkCommand; + state.scopePaths = cloneStringArray(configEntry.scopePaths); + state.offLimits = cloneStringArray(configEntry.offLimits); + state.constraints = cloneStringArray(configEntry.constraints); + state.segmentFingerprint = + typeof configEntry.segmentFingerprint === "string" ? configEntry.segmentFingerprint : null; + state.secondaryMetrics = hydrateMetricDefs(configEntry.secondaryMetrics); continue; } if (!isRunEntry(parsed)) continue; const result: ExperimentResult = { + runNumber: typeof parsed.run === "number" && Number.isFinite(parsed.run) ? parsed.run : null, commit: typeof parsed.commit === "string" ? parsed.commit : "", metric: typeof parsed.metric === "number" && Number.isFinite(parsed.metric) ? parsed.metric : 0, metrics: cloneNumericMetrics(parsed.metrics), @@ -195,17 +234,19 @@ export function reconstructStateFromJsonl(workDir: string): ReconstructedExperim export function reconstructControlState(entries: SessionEntry[]): ReconstructedControlState { let autoresearchMode = false; let goal: string | null = null; + let lastMode: ReconstructedControlState["lastMode"] = null; for (const entry of entries) { if (entry.type !== "custom" || entry.customType !== "autoresearch-control") continue; const data = parseControlEntry(entry.data); if (!data) continue; + lastMode = data.mode; autoresearchMode = data.mode === "on"; goal = data.goal ?? goal; if (data.mode === "clear") { goal = null; } } - return { autoresearchMode, goal }; + return { autoresearchMode, goal, lastMode }; } export function createRuntimeStore(): RuntimeStore { @@ -240,6 +281,51 @@ function isConfigEntry(value: unknown): value is AutoresearchJsonConfigEntry { return candidate.type === "config"; } +function parseConfigEntry(value: unknown): AutoresearchJsonConfigEntry | null { + if (!isConfigEntry(value)) return null; + const candidate = value as AutoresearchJsonConfigEntry; + const config: AutoresearchJsonConfigEntry = { type: "config" }; + if (typeof candidate.name === "string" && candidate.name.trim().length > 0) { + config.name = candidate.name; + } + if (typeof candidate.metricName === "string" && candidate.metricName.trim().length > 0) { + config.metricName = candidate.metricName; + } + if (typeof candidate.metricUnit === "string") { + config.metricUnit = candidate.metricUnit; + } + if (candidate.bestDirection === "lower" || candidate.bestDirection === "higher") { + config.bestDirection = candidate.bestDirection; + } + if (typeof candidate.benchmarkCommand === "string" && candidate.benchmarkCommand.trim().length > 0) { + config.benchmarkCommand = candidate.benchmarkCommand; + } + if (Array.isArray(candidate.secondaryMetrics)) { + config.secondaryMetrics = normalizeAutoresearchList( + candidate.secondaryMetrics.filter((item): item is string => typeof item === "string"), + ); + } + if (Array.isArray(candidate.scopePaths)) { + config.scopePaths = normalizeAutoresearchList( + candidate.scopePaths.filter((item): item is string => typeof item === "string").map(normalizeContractPathSpec), + ); + } + if (Array.isArray(candidate.offLimits)) { + config.offLimits = normalizeAutoresearchList( + candidate.offLimits.filter((item): item is string => typeof item === "string").map(normalizeContractPathSpec), + ); + } + if (Array.isArray(candidate.constraints)) { + config.constraints = normalizeAutoresearchList( + candidate.constraints.filter((item): item is string => typeof item === "string"), + ); + } + if (typeof candidate.segmentFingerprint === "string" && candidate.segmentFingerprint.trim().length > 0) { + config.segmentFingerprint = candidate.segmentFingerprint; + } + return config; +} + function isRunEntry(value: unknown): value is AutoresearchJsonRunEntry { if (typeof value !== "object" || value === null) return false; const candidate = value as { type?: unknown }; @@ -262,6 +348,19 @@ function cloneNumericMetrics(value: unknown): NumericMetricMap { return clone; } +function cloneStringArray(value: unknown): string[] { + if (!Array.isArray(value)) return []; + return value.filter((item): item is string => typeof item === "string"); +} + +function hydrateMetricDefs(metricNames: string[] | undefined): MetricDef[] { + if (!metricNames) return []; + return metricNames.map(name => ({ + name, + unit: inferMetricUnitFromName(name), + })); +} + function cloneAsi(value: unknown): ExperimentResult["asi"] { if (typeof value !== "object" || value === null) return undefined; return structuredClone(value) as ExperimentResult["asi"]; diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts index 19744fd2d..5e81714ba 100644 --- a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -5,7 +5,21 @@ import { Text } from "@oh-my-pi/pi-tui"; import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; -import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "../helpers"; +import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; +import { + buildAutoresearchSegmentFingerprint, + contractListsEqual, + contractPathListsEqual, + loadAutoresearchScriptSnapshot, + readAutoresearchContract, +} from "../contract"; +import { + inferMetricUnitFromName, + isAutoresearchShCommand, + readMaxExperiments, + resolveWorkDir, + validateWorkDir, +} from "../helpers"; import { cloneExperimentState } from "../state"; import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types"; @@ -26,6 +40,23 @@ const initExperimentSchema = Type.Object({ description: "Whether lower or higher values are better. Defaults to lower.", }), ), + benchmark_command: Type.String({ + description: "Benchmark command recorded in autoresearch.md.", + }), + scope_paths: Type.Array(Type.String(), { + description: "Files in Scope from autoresearch.md. Must be non-empty.", + minItems: 1, + }), + off_limits: Type.Optional( + Type.Array(Type.String(), { + description: "Off Limits paths from autoresearch.md.", + }), + ), + constraints: Type.Optional( + Type.Array(Type.String(), { + description: "Constraints from autoresearch.md.", + }), + ), }); interface InitExperimentDetails { @@ -53,6 +84,120 @@ export function createInitExperimentTool( const runtime = options.getRuntime(ctx); const state = runtime.state; const isReinitializing = state.results.length > 0; + const workDir = resolveWorkDir(ctx.cwd); + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + const errors = [...contractResult.errors, ...scriptSnapshot.errors]; + if (errors.length > 0) { + return { + content: [{ type: "text", text: `Error: ${errors.join(" ")}` }], + }; + } + + const benchmarkContract = contractResult.contract.benchmark; + const expectedDirection = benchmarkContract.direction ?? "lower"; + const expectedMetricUnit = benchmarkContract.metricUnit; + if (benchmarkContract.command && !isAutoresearchShCommand(benchmarkContract.command)) { + return { + content: [ + { + type: "text", + text: + "Error: Benchmark.command in autoresearch.md must invoke `autoresearch.sh` directly. " + + "Move the real workload into `autoresearch.sh` and re-run init_experiment.", + }, + ], + }; + } + if (benchmarkContract.command !== params.benchmark_command.trim()) { + return { + content: [ + { + type: "text", + text: + "Error: benchmark_command does not match autoresearch.md. " + + `Expected: ${benchmarkContract.command ?? "(missing)"}\nReceived: ${params.benchmark_command}`, + }, + ], + }; + } + if (benchmarkContract.primaryMetric !== params.metric_name.trim()) { + return { + content: [ + { + type: "text", + text: + "Error: metric_name does not match autoresearch.md. " + + `Expected: ${benchmarkContract.primaryMetric ?? "(missing)"}\nReceived: ${params.metric_name}`, + }, + ], + }; + } + if ((params.metric_unit ?? "") !== expectedMetricUnit) { + return { + content: [ + { + type: "text", + text: + "Error: metric_unit does not match autoresearch.md. " + + `Expected: ${expectedMetricUnit || "(empty)"}\nReceived: ${params.metric_unit ?? "(empty)"}`, + }, + ], + }; + } + if ((params.direction ?? "lower") !== expectedDirection) { + return { + content: [ + { + type: "text", + text: + "Error: direction does not match autoresearch.md. " + + `Expected: ${expectedDirection}\nReceived: ${params.direction ?? "lower"}`, + }, + ], + }; + } + if (!contractPathListsEqual(params.scope_paths, contractResult.contract.scopePaths)) { + return { + content: [ + { + type: "text", + text: + "Error: scope_paths do not match autoresearch.md. " + + `Expected: ${contractResult.contract.scopePaths.join(", ")}`, + }, + ], + }; + } + if (!contractPathListsEqual(params.off_limits ?? [], contractResult.contract.offLimits)) { + return { + content: [ + { + type: "text", + text: + "Error: off_limits do not match autoresearch.md. " + + `Expected: ${contractResult.contract.offLimits.join(", ") || "(empty)"}`, + }, + ], + }; + } + if (!contractListsEqual(params.constraints ?? [], contractResult.contract.constraints)) { + return { + content: [ + { + type: "text", + text: + "Error: constraints do not match autoresearch.md. " + + `Expected: ${contractResult.contract.constraints.join(", ") || "(empty)"}`, + }, + ], + }; + } + + const segmentFingerprint = buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); state.name = params.name; state.metricName = params.metric_name; @@ -61,12 +206,19 @@ export function createInitExperimentTool( state.maxExperiments = readMaxExperiments(ctx.cwd); state.bestMetric = null; state.confidence = null; - state.secondaryMetrics = []; + state.secondaryMetrics = benchmarkContract.secondaryMetrics.map(name => ({ + name, + unit: inferMetricUnitFromName(name), + })); + state.benchmarkCommand = params.benchmark_command.trim(); + state.scopePaths = [...contractResult.contract.scopePaths]; + state.offLimits = [...contractResult.contract.offLimits]; + state.constraints = [...contractResult.contract.constraints]; + state.segmentFingerprint = segmentFingerprint; if (isReinitializing) { state.currentSegment += 1; } - const workDir = resolveWorkDir(ctx.cwd); const jsonlPath = path.join(workDir, "autoresearch.jsonl"); const configLine = JSON.stringify({ type: "config", @@ -74,6 +226,12 @@ export function createInitExperimentTool( metricName: state.metricName, metricUnit: state.metricUnit, bestDirection: state.bestDirection, + benchmarkCommand: state.benchmarkCommand, + secondaryMetrics: state.secondaryMetrics.map(metric => metric.name), + scopePaths: state.scopePaths, + offLimits: state.offLimits, + constraints: state.constraints, + segmentFingerprint, }); if (isReinitializing) { @@ -89,7 +247,9 @@ export function createInitExperimentTool( const lines = [ `Experiment initialized: ${state.name}`, `Metric: ${state.metricName} (${state.metricUnit || "unitless"}, ${state.bestDirection} is better)`, + `Benchmark command: ${state.benchmarkCommand}`, `Working directory: ${workDir}`, + `Files in Scope: ${state.scopePaths.join(", ")}`, isReinitializing ? "Previous results remain in history. This starts a new segment and requires a fresh baseline." : "Now run the baseline experiment and log it.", @@ -107,12 +267,12 @@ export function createInitExperimentTool( return new Text(renderInitCall(args.name, theme), 0, 0); }, renderResult(result): Text { - const text = result.content.find(part => part.type === "text")?.text ?? ""; + const text = replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""); return new Text(text, 0, 0); }, }; } function renderInitCall(name: string, theme: Theme): string { - return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", name)}`; + return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", truncateToWidth(replaceTabs(name), 100))}`; } diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index 2a7420a46..1a25db2ef 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -2,14 +2,22 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { StringEnum } from "@oh-my-pi/pi-ai"; import { Text } from "@oh-my-pi/pi-tui"; +import { logger } from "@oh-my-pi/pi-utils"; import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; +import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; +import { getAutoresearchFingerprintMismatchError, pathMatchesContractPath } from "../contract"; +import { getCurrentAutoresearchBranch, parseWorkDirDirtyPaths } from "../git"; import { + AUTORESEARCH_COMMITTABLE_FILES, formatNum, inferMetricUnitFromName, + isAutoresearchCommittableFile, + isAutoresearchLocalStatePath, + isBetter, mergeAsi, - PROTECTED_AUTORESEARCH_FILES, + readPendingRunSummary, resolveWorkDir, validateWorkDir, } from "../helpers"; @@ -19,6 +27,7 @@ import { currentResults, findBaselineMetric, findBaselineSecondary, + findBestKeptMetric, } from "../state"; import type { ASIData, @@ -29,6 +38,8 @@ import type { NumericMetricMap, } from "../types"; +const EXPERIMENT_TOOL_NAMES = ["init_experiment", "run_experiment", "log_experiment"]; + const logExperimentSchema = Type.Object({ commit: Type.String({ description: "Current git commit hash or placeholder.", @@ -64,6 +75,11 @@ interface PreservedFile { path: string; } +interface KeepCommitResult { + error?: string; + note?: string; +} + export function createLogExperimentTool( options: AutoresearchToolFactoryOptions, ): ToolDefinition { @@ -85,7 +101,55 @@ export function createLogExperimentTool( const runtime = options.getRuntime(ctx); const state = runtime.state; const workDir = resolveWorkDir(ctx.cwd); - const secondaryMetrics = cloneMetrics(params.metrics); + const fingerprintError = getAutoresearchFingerprintMismatchError(state.segmentFingerprint, workDir); + if (fingerprintError) { + return { + content: [{ type: "text", text: `Error: ${fingerprintError}` }], + }; + } + + const pendingRun = + runtime.lastRunSummary ?? (await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results))); + if (!pendingRun) { + return { + content: [{ type: "text", text: "Error: no unlogged run is available. Run run_experiment first." }], + }; + } + runtime.lastRunSummary = pendingRun; + runtime.lastRunAsi = pendingRun.parsedAsi; + runtime.lastRunChecks = + pendingRun.checksPass === null + ? null + : { + pass: pendingRun.checksPass, + output: "", + duration: pendingRun.checksDurationSeconds ?? 0, + }; + runtime.lastRunDuration = pendingRun.durationSeconds; + + if (pendingRun.parsedPrimary !== null && params.metric !== pendingRun.parsedPrimary) { + return { + content: [ + { + type: "text", + text: + "Error: metric does not match the parsed primary metric from the pending run.\n" + + `Expected: ${pendingRun.parsedPrimary}\nReceived: ${params.metric}`, + }, + ], + }; + } + + if (params.status === "keep" && !pendingRun.passed) { + return { + content: [ + { + type: "text", + text: "Error: cannot keep this run because the pending benchmark did not pass. Log it as crash or checks_failed instead.", + }, + ], + }; + } if (params.status === "keep" && runtime.lastRunChecks && !runtime.lastRunChecks.pass) { return { @@ -98,6 +162,14 @@ export function createLogExperimentTool( }; } + const observedStatusError = validateObservedStatus(params.status, pendingRun); + if (observedStatusError) { + return { + content: [{ type: "text", text: `Error: ${observedStatusError}` }], + }; + } + + const secondaryMetrics = buildSecondaryMetrics(params.metrics, pendingRun.parsedMetrics, state.metricName); const validationError = validateSecondaryMetrics(state, secondaryMetrics, params.force ?? false); if (validationError) { return { @@ -112,7 +184,37 @@ export function createLogExperimentTool( content: [{ type: "text", text: `Error: ${asiValidationError}` }], }; } + + let keepScopeValidation: { committablePaths: string[] } | undefined; + if (params.status === "keep") { + const scopeValidation = await validateKeepPaths(options, workDir, state); + if (typeof scopeValidation === "string") { + return { + content: [{ type: "text", text: `Error: ${scopeValidation}` }], + }; + } + const currentBestMetric = findBestKeptMetric(state.results, state.currentSegment, state.bestDirection); + if ( + currentBestMetric !== null && + params.metric !== currentBestMetric && + !isBetter(params.metric, currentBestMetric, state.bestDirection) + ) { + return { + content: [ + { + type: "text", + text: + "Error: cannot keep this run because the primary metric regressed.\n" + + `Current best: ${currentBestMetric}\nReceived: ${params.metric}`, + }, + ], + }; + } + keepScopeValidation = scopeValidation; + } + const experiment: ExperimentResult = { + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, commit: params.commit.slice(0, 7), metric: params.metric, metrics: secondaryMetrics, @@ -124,32 +226,96 @@ export function createLogExperimentTool( asi: mergedAsi, }; + const activeBranch = await getCurrentAutoresearchBranch(options.pi, workDir); + if (!activeBranch) { + return { + content: [ + { + type: "text", + text: + "Error: autoresearch keep/discard actions require an active `autoresearch/...` branch. " + + "Run `/autoresearch` again to restore the protected branch before logging this run.", + }, + ], + }; + } + + let gitNote: string | null = null; + if (params.status === "keep") { + const commitResult = await commitKeptExperiment(options, workDir, state, experiment, keepScopeValidation); + if (commitResult.error) { + return { + content: [{ type: "text", text: `Error: ${commitResult.error}` }], + }; + } + gitNote = commitResult.note ?? null; + } else { + const revertResult = await revertFailedExperiment(options, workDir); + if (revertResult.error) { + return { + content: [{ type: "text", text: `Error: ${revertResult.error}` }], + }; + } + gitNote = revertResult.note ?? null; + } + + const previousState = cloneExperimentState(state); state.results.push(experiment); - runtime.experimentsThisSession += 1; registerSecondaryMetrics(state, secondaryMetrics); state.bestMetric = findBaselineMetric(state.results, state.currentSegment); state.confidence = computeConfidence(state.results, state.currentSegment, state.bestDirection); experiment.confidence = state.confidence; - persistRun(workDir, state.results.length, experiment); - - let gitNote: string | null = null; - if (params.status === "keep") { - gitNote = await commitKeptExperiment(options, workDir, state, experiment); - } else { - gitNote = await revertFailedExperiment(options, workDir); + const wallClockSeconds = runtime.lastRunDuration; + try { + persistRun(workDir, experiment); + } catch (error) { + runtime.state = previousState; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + throw error; + } + try { + await updateRunMetadata(runtime.lastRunArtifactDir ?? pendingRun.runDirectory, { + commit: experiment.commit, + confidence: experiment.confidence, + description: experiment.description, + gitNote, + loggedAt: new Date(experiment.timestamp).toISOString(), + loggedAsi: experiment.asi, + loggedMetric: experiment.metric, + loggedMetrics: experiment.metrics, + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, + status: experiment.status, + wallClockSeconds, + }); + } catch (error) { + logger.warn("Failed to update autoresearch run metadata after persisting JSONL history", { + error: error instanceof Error ? error.message : String(error), + runDirectory: runtime.lastRunArtifactDir ?? pendingRun.runDirectory, + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, + }); } - const wallClockSeconds = runtime.lastRunDuration; runtime.runningExperiment = null; runtime.lastRunChecks = null; runtime.lastRunDuration = null; runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = null; + runtime.lastRunNumber = null; + runtime.lastRunSummary = null; const currentSegmentRuns = currentResults(state.results, state.currentSegment).length; const text = buildLogText(state, experiment, currentSegmentRuns, wallClockSeconds, gitNote); if (state.maxExperiments !== null && currentSegmentRuns >= state.maxExperiments) { runtime.autoresearchMode = false; + options.pi.appendEntry( + "autoresearch-control", + runtime.goal ? { mode: "off", goal: runtime.goal } : { mode: "off" }, + ); + await options.pi.setActiveTools( + options.pi.getActiveTools().filter(name => !EXPERIMENT_TOOL_NAMES.includes(name)), + ); } options.dashboard.updateWidget(ctx, runtime); options.dashboard.requestRender(); @@ -169,8 +335,9 @@ export function createLogExperimentTool( }, renderCall(args, _options, theme): Text { const color = args.status === "keep" ? "success" : args.status === "discard" ? "warning" : "error"; + const description = truncateToWidth(replaceTabs(args.description), 100); return new Text( - `${theme.fg("toolTitle", theme.bold("log_experiment"))} ${theme.fg(color, args.status)} ${theme.fg("muted", args.description)}`, + `${theme.fg("toolTitle", theme.bold("log_experiment"))} ${theme.fg(color, args.status)} ${theme.fg("muted", description)}`, 0, 0, ); @@ -178,7 +345,7 @@ export function createLogExperimentTool( renderResult(result, _options, theme): Text { const details = result.details; if (!details) { - return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + return new Text(replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""), 0, 0); } const summary = renderSummary(details, theme); return new Text(summary, 0, 0); @@ -190,6 +357,22 @@ function cloneMetrics(value: NumericMetricMap | undefined): NumericMetricMap { return value ? { ...value } : {}; } +function buildSecondaryMetrics( + overrides: NumericMetricMap | undefined, + parsedMetrics: NumericMetricMap | null, + primaryMetricName: string, +): NumericMetricMap { + const merged: NumericMetricMap = {}; + for (const [name, value] of Object.entries(parsedMetrics ?? {})) { + if (name === primaryMetricName) continue; + merged[name] = value; + } + for (const [name, value] of Object.entries(cloneMetrics(overrides))) { + merged[name] = value; + } + return merged; +} + function sanitizeAsi(value: { [key: string]: unknown } | undefined): ASIData | undefined { if (!value) return undefined; const result: ASIData = {}; @@ -269,29 +452,71 @@ function registerSecondaryMetrics(state: ExperimentState, metrics: NumericMetric } } -function persistRun(workDir: string, runNumber: number, experiment: ExperimentResult): void { +function persistRun(workDir: string, experiment: ExperimentResult): void { const entry = { - run: runNumber, + run: experiment.runNumber, ...experiment, }; const jsonlPath = path.join(workDir, "autoresearch.jsonl"); fs.appendFileSync(jsonlPath, `${JSON.stringify(entry)}\n`); } +function collectLoggedRunNumbers(results: ExperimentResult[]): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} + +function validateObservedStatus( + status: ExperimentResult["status"], + pendingRun: { checksPass: boolean | null; passed: boolean }, +): string | null { + if (pendingRun.checksPass === false) { + return status === "checks_failed" + ? null + : "benchmark checks failed for the pending run. Log it as checks_failed."; + } + if (!pendingRun.passed) { + return status === "crash" ? null : "the pending benchmark failed. Log it as crash."; + } + return status === "keep" || status === "discard" ? null : "the pending benchmark passed. Log it as keep or discard."; +} + async function commitKeptExperiment( options: AutoresearchToolFactoryOptions, workDir: string, state: ExperimentState, experiment: ExperimentResult, -): Promise { - const addResult = await options.pi.exec("git", ["add", "-A"], { cwd: workDir, timeout: 10_000 }); - if (addResult.code !== 0) { - return `git add failed: ${mergeStdoutStderr(addResult).trim() || `exit ${addResult.code}`}`; + scopeValidation: { committablePaths: string[] } | undefined, +): Promise { + if (!scopeValidation || scopeValidation.committablePaths.length === 0) { + return { note: "nothing to commit" }; } - const diffResult = await options.pi.exec("git", ["diff", "--cached", "--quiet"], { cwd: workDir, timeout: 10_000 }); + const addResult = await options.pi.exec("git", ["add", "--all", "--", ...scopeValidation.committablePaths], { + cwd: workDir, + timeout: 10_000, + }); + if (addResult.code !== 0) { + return { + error: `git add failed: ${mergeStdoutStderr(addResult).trim() || `exit ${addResult.code}`}`, + }; + } + + const diffResult = await options.pi.exec( + "git", + ["diff", "--cached", "--quiet", "--", ...scopeValidation.committablePaths], + { + cwd: workDir, + timeout: 10_000, + }, + ); if (diffResult.code === 0) { - return "nothing to commit"; + return { note: "nothing to commit" }; } const payload: { [key: string]: string | number } = { @@ -302,12 +527,18 @@ async function commitKeptExperiment( payload[name] = value; } const commitMessage = `${experiment.description}\n\nResult: ${JSON.stringify(payload)}`; - const commitResult = await options.pi.exec("git", ["commit", "-m", commitMessage], { - cwd: workDir, - timeout: 10_000, - }); + const commitResult = await options.pi.exec( + "git", + ["commit", "-m", commitMessage, "--", ...scopeValidation.committablePaths], + { + cwd: workDir, + timeout: 10_000, + }, + ); if (commitResult.code !== 0) { - return `git commit failed: ${mergeStdoutStderr(commitResult).trim() || `exit ${commitResult.code}`}`; + return { + error: `git commit failed: ${mergeStdoutStderr(commitResult).trim() || `exit ${commitResult.code}`}`, + }; } const revParseResult = await options.pi.exec("git", ["rev-parse", "--short=7", "HEAD"], { @@ -322,28 +553,58 @@ async function commitKeptExperiment( mergeStdoutStderr(commitResult) .split("\n") .find(line => line.trim().length > 0) ?? "committed"; - return summaryLine.trim(); + return { note: summaryLine.trim() }; } -async function revertFailedExperiment(options: AutoresearchToolFactoryOptions, workDir: string): Promise { +async function revertFailedExperiment( + options: AutoresearchToolFactoryOptions, + workDir: string, +): Promise { const preservedFiles = preserveAutoresearchFiles(workDir); - const resetResult = await options.pi.exec("git", ["reset", "--hard", "HEAD"], { cwd: workDir, timeout: 10_000 }); - const cleanResult = await options.pi.exec("git", ["clean", "-fd"], { cwd: workDir, timeout: 10_000 }); + const restoreResult = await options.pi.exec( + "git", + ["restore", "--source=HEAD", "--staged", "--worktree", "--", "."], + { cwd: workDir, timeout: 10_000 }, + ); + const cleanResult = await options.pi.exec("git", ["clean", "-fd", "--", "."], { cwd: workDir, timeout: 10_000 }); restoreAutoresearchFiles(preservedFiles); - - const notes: string[] = ["reverted changes"]; - if (resetResult.code !== 0) { - notes.push(`git reset failed: ${mergeStdoutStderr(resetResult).trim() || `exit ${resetResult.code}`}`); + if (restoreResult.code !== 0) { + return { + error: `git restore failed: ${mergeStdoutStderr(restoreResult).trim() || `exit ${restoreResult.code}`}`, + }; } if (cleanResult.code !== 0) { - notes.push(`git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`); + return { + error: `git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`, + }; } - return notes.join("; "); + const dirtyCheckResult = await options.pi.exec( + "git", + ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."], + { cwd: workDir, timeout: 10_000 }, + ); + if (dirtyCheckResult.code !== 0) { + return { + error: `git status failed after cleanup: ${mergeStdoutStderr(dirtyCheckResult).trim() || `exit ${dirtyCheckResult.code}`}`, + }; + } + const workDirPrefix = await readGitWorkDirPrefix(options, workDir); + const remainingDirtyPaths = parseWorkDirDirtyPaths(dirtyCheckResult.stdout, workDirPrefix).filter( + relativePath => !isAutoresearchLocalStatePath(relativePath), + ); + if (remainingDirtyPaths.length > 0) { + return { + error: + "Autoresearch cleanup left the worktree dirty. Resolve these paths before continuing: " + + remainingDirtyPaths.join(", "), + }; + } + return { note: "reverted changes" }; } function preserveAutoresearchFiles(workDir: string): PreservedFile[] { const files: PreservedFile[] = []; - for (const relativePath of PROTECTED_AUTORESEARCH_FILES) { + for (const relativePath of [...AUTORESEARCH_COMMITTABLE_FILES, "autoresearch.jsonl"]) { const absolutePath = path.join(workDir, relativePath); if (!fs.existsSync(absolutePath)) continue; files.push({ @@ -351,6 +612,10 @@ function preserveAutoresearchFiles(workDir: string): PreservedFile[] { path: absolutePath, }); } + const localStateDir = path.join(workDir, ".autoresearch"); + if (fs.existsSync(localStateDir)) { + collectDirectoryFiles(localStateDir, files); + } return files; } @@ -365,6 +630,110 @@ function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { return `${result.stdout}${result.stderr}`; } +async function validateKeepPaths( + options: AutoresearchToolFactoryOptions, + workDir: string, + state: ExperimentState, +): Promise<{ committablePaths: string[] } | string> { + if (state.scopePaths.length === 0) { + return "Files in Scope is empty for the current segment. Re-run init_experiment after fixing autoresearch.md."; + } + + const statusResult = await options.pi.exec( + "git", + ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."], + { + cwd: workDir, + timeout: 10_000, + }, + ); + if (statusResult.code !== 0) { + return `git status failed: ${mergeStdoutStderr(statusResult).trim() || `exit ${statusResult.code}`}`; + } + + const workDirPrefix = await readGitWorkDirPrefix(options, workDir); + const committablePaths: string[] = []; + for (const normalizedPath of parseWorkDirDirtyPaths(statusResult.stdout, workDirPrefix)) { + if (isAutoresearchLocalStatePath(normalizedPath)) { + continue; + } + if (isAutoresearchCommittableFile(normalizedPath)) { + committablePaths.push(normalizedPath); + continue; + } + if (state.offLimits.some(spec => pathMatchesContractPath(normalizedPath, spec))) { + return `cannot keep this run because ${normalizedPath} is listed under Off Limits in autoresearch.md`; + } + if (!state.scopePaths.some(spec => pathMatchesContractPath(normalizedPath, spec))) { + return `cannot keep this run because ${normalizedPath} is outside Files in Scope`; + } + committablePaths.push(normalizedPath); + } + + return { committablePaths }; +} + +function collectDirectoryFiles(directory: string, files: PreservedFile[]): void { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const absolutePath = path.join(directory, entry.name); + if (entry.isDirectory()) { + collectDirectoryFiles(absolutePath, files); + continue; + } + files.push({ + content: fs.readFileSync(absolutePath), + path: absolutePath, + }); + } +} + +async function updateRunMetadata( + runDirectory: string | null, + metadata: { + commit: string; + confidence: number | null; + description: string; + gitNote: string | null; + loggedAt: string; + loggedAsi: ASIData | undefined; + loggedMetric: number; + loggedMetrics: NumericMetricMap; + runNumber: number | null; + status: ExperimentResult["status"]; + wallClockSeconds: number | null; + }, +): Promise { + if (!runDirectory) return; + const runJsonPath = path.join(runDirectory, "run.json"); + let existing: Record = {}; + try { + existing = (await Bun.file(runJsonPath).json()) as Record; + } catch { + existing = {}; + } + await Bun.write( + runJsonPath, + JSON.stringify( + { + ...existing, + loggedRunNumber: metadata.runNumber, + loggedAt: metadata.loggedAt, + loggedAsi: metadata.loggedAsi, + loggedMetric: metadata.loggedMetric, + loggedMetrics: metadata.loggedMetrics, + status: metadata.status, + description: metadata.description, + commit: metadata.commit, + gitNote: metadata.gitNote, + confidence: metadata.confidence, + wallClockSeconds: metadata.wallClockSeconds, + }, + null, + 2, + ), + ); +} + function buildLogText( state: ExperimentState, experiment: ExperimentResult, @@ -372,7 +741,8 @@ function buildLogText( wallClockSeconds: number | null, gitNote: string | null, ): string { - const lines = [`Logged run #${state.results.length}: ${experiment.status} - ${experiment.description}`]; + const displayRunNumber = experiment.runNumber ?? state.results.length; + const lines = [`Logged run #${displayRunNumber}: ${experiment.status} - ${experiment.description}`]; if (wallClockSeconds !== null) { lines.push(`Wall clock: ${wallClockSeconds.toFixed(1)}s`); } @@ -422,6 +792,12 @@ function buildLogText( return lines.join("\n"); } +async function readGitWorkDirPrefix(options: AutoresearchToolFactoryOptions, workDir: string): Promise { + const prefixResult = await options.pi.exec("git", ["rev-parse", "--show-prefix"], { cwd: workDir, timeout: 5_000 }); + if (prefixResult.code !== 0) return ""; + return prefixResult.stdout.trim(); +} + function truncateAsiValue(value: ASIData[string]): string { const text = typeof value === "string" ? value : JSON.stringify(value); return text.length > 120 ? `${text.slice(0, 117)}...` : text; @@ -430,7 +806,7 @@ function truncateAsiValue(value: ASIData[string]): string { function renderSummary(details: LogDetails, theme: Theme): string { const { experiment, state } = details; const color = experiment.status === "keep" ? "success" : experiment.status === "discard" ? "warning" : "error"; - let summary = `${theme.fg(color, experiment.status.toUpperCase())} ${theme.fg("muted", experiment.description)}`; + let summary = `${theme.fg(color, experiment.status.toUpperCase())} ${theme.fg("muted", truncateToWidth(replaceTabs(experiment.description), 100))}`; summary += ` ${theme.fg("accent", `${state.metricName}=${formatNum(experiment.metric, state.metricUnit)}`)}`; if (state.bestMetric !== null) { summary += ` ${theme.fg("dim", `baseline ${formatNum(state.bestMetric, state.metricUnit)}`)}`; diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts index a5e502b15..a393c3647 100644 --- a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -7,16 +7,20 @@ import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output"; +import { replaceTabs, shortenPath, truncateToWidth } from "../../tools/render-utils"; +import { getAutoresearchFingerprintMismatchError } from "../contract"; import { - createTempFileAllocator, EXPERIMENT_MAX_BYTES, EXPERIMENT_MAX_LINES, formatElapsed, formatNum, + getAutoresearchRunDirectory, + getNextAutoresearchRunNumber, isAutoresearchShCommand, killTree, parseAsiLines, parseMetricLines, + readPendingRunSummary, resolveWorkDir, validateWorkDir, } from "../helpers"; @@ -39,19 +43,27 @@ const runExperimentSchema = Type.Object({ }); interface ProcessExecutionResult { - actualTotalBytes: number; exitCode: number | null; killed: boolean; + logPath: string; output: string; - tempFilePath?: string; } interface ChecksExecutionResult { code: number | null; killed: boolean; + logPath: string; output: string; } +interface ProgressSnapshot { + elapsed: string; + runDirectory: string; + fullOutputPath: string; + tailOutput: string; + truncation?: RunExperimentProgressDetails["truncation"]; +} + export function createRunExperimentTool( options: AutoresearchToolFactoryOptions, ): ToolDefinition { @@ -59,7 +71,7 @@ export function createRunExperimentTool( name: "run_experiment", label: "Run Experiment", description: - "Run an experiment command with timing, tail capture, structured metric parsing, and optional autoresearch.checks.sh validation.", + "Run an experiment command with timing, output capture, structured metric parsing, durable run artifacts, and optional autoresearch.checks.sh validation.", parameters: runExperimentSchema, defaultInactive: true, async execute(_toolCallId, params, signal, onUpdate, ctx) { @@ -75,6 +87,25 @@ export function createRunExperimentTool( const workDir = resolveWorkDir(ctx.cwd); const checksPath = path.join(workDir, "autoresearch.checks.sh"); const autoresearchScriptPath = path.join(workDir, "autoresearch.sh"); + const fingerprintError = getAutoresearchFingerprintMismatchError(state.segmentFingerprint, workDir); + if (fingerprintError) { + return { + content: [{ type: "text", text: `Error: ${fingerprintError}` }], + }; + } + + if (state.benchmarkCommand && params.command.trim() !== state.benchmarkCommand) { + return { + content: [ + { + type: "text", + text: + "Error: command does not match the benchmark command recorded for this segment.\n" + + `Expected: ${state.benchmarkCommand}\nReceived: ${params.command}`, + }, + ], + }; + } if (fs.existsSync(autoresearchScriptPath) && !isAutoresearchShCommand(params.command)) { return { @@ -104,9 +135,54 @@ export function createRunExperimentTool( } } + const pendingRun = + runtime.lastRunSummary ?? (await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results))); + if (pendingRun) { + return { + content: [ + { + type: "text", + text: + `Error: run #${pendingRun.runNumber} has not been logged yet. ` + + "Call log_experiment before starting another benchmark run.", + }, + ], + }; + } + + const runNumber = getNextAutoresearchRunNumber(workDir, runtime.lastRunNumber); + const runDirectory = getAutoresearchRunDirectory(workDir, runNumber); + const benchmarkLogPath = path.join(runDirectory, "benchmark.log"); + const checksLogPath = path.join(runDirectory, "checks.log"); + const runJsonPath = path.join(runDirectory, "run.json"); + await fs.promises.mkdir(runDirectory, { recursive: true }); + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = runDirectory; + runtime.lastRunNumber = runNumber; + runtime.lastRunSummary = null; + await Bun.write( + runJsonPath, + JSON.stringify( + { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath, + command: params.command, + startedAt: new Date().toISOString(), + }, + null, + 2, + ), + ); + runtime.runningExperiment = { startedAt: Date.now(), command: params.command, + runDirectory, + runNumber, }; options.dashboard.updateWidget(ctx, runtime); options.dashboard.requestRender(); @@ -116,8 +192,9 @@ export function createRunExperimentTool( let execution: ProcessExecutionResult; try { execution = await executeProcess({ - command: params.command, + command: ["bash", "-lc", params.command], cwd: workDir, + logPath: benchmarkLogPath, timeoutMs, signal, onProgress: details => { @@ -128,6 +205,7 @@ export function createRunExperimentTool( elapsed: details.elapsed, truncation: details.truncation, fullOutputPath: details.fullOutputPath, + runDirectory: details.runDirectory, }, }); }, @@ -146,12 +224,14 @@ export function createRunExperimentTool( let checksTimedOut = false; let checksOutput = ""; let checksDuration = 0; + let checksLogPathValue: string | undefined; if (benchmarkPassed && fs.existsSync(checksPath)) { const checksStartedAt = Date.now(); - const checksResult = runChecks({ + const checksResult = await runChecks({ cwd: workDir, pathToChecks: checksPath, + logPath: checksLogPath, timeoutMs: Math.max(0, Math.floor((params.checks_timeout_seconds ?? 300) * 1000)), signal, }); @@ -159,6 +239,7 @@ export function createRunExperimentTool( checksTimedOut = checksResult.killed; checksPass = checksResult.code === 0 && !checksResult.killed; checksOutput = checksResult.output; + checksLogPathValue = checksResult.logPath; } runtime.lastRunChecks = @@ -179,12 +260,6 @@ export function createRunExperimentTool( maxLines: DEFAULT_MAX_LINES, }); - let fullOutputPath = execution.tempFilePath; - if (!fullOutputPath && llmTruncation.truncated) { - fullOutputPath = createTempFileAllocator()(); - fs.writeFileSync(fullOutputPath, execution.output); - } - const parsedMetricsMap = parseMetricLines(execution.output); const parsedMetrics = parsedMetricsMap.size > 0 ? Object.fromEntries(parsedMetricsMap.entries()) : null; const parsedPrimary = parsedMetricsMap.get(state.metricName) ?? null; @@ -192,6 +267,10 @@ export function createRunExperimentTool( runtime.lastRunAsi = parsedAsi; const resultDetails: RunDetails = { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath: checksLogPathValue, command: params.command, exitCode: execution.exitCode, durationSeconds, @@ -209,8 +288,50 @@ export function createRunExperimentTool( metricName: state.metricName, metricUnit: state.metricUnit, truncation: llmTruncation.truncated ? llmTruncation : undefined, - fullOutputPath, + fullOutputPath: execution.logPath, }; + runtime.lastRunSummary = { + checksDurationSeconds: checksDuration, + checksPass, + checksTimedOut, + command: params.command, + durationSeconds, + parsedAsi, + parsedMetrics, + parsedPrimary, + passed: resultDetails.passed, + runDirectory, + runNumber, + }; + + await Bun.write( + runJsonPath, + JSON.stringify( + { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath: checksLogPathValue, + command: params.command, + completedAt: new Date().toISOString(), + durationSeconds, + exitCode: execution.exitCode, + timedOut: execution.killed, + checks: { + durationSeconds: checksDuration, + passed: checksPass, + timedOut: checksTimedOut, + }, + parsedMetrics, + parsedPrimary, + parsedAsi, + truncation: resultDetails.truncation, + fullOutputPath: resultDetails.fullOutputPath, + }, + null, + 2, + ), + ); return { content: [{ type: "text", text: buildRunText(resultDetails, llmTruncation.content, state.bestMetric) }], @@ -218,8 +339,9 @@ export function createRunExperimentTool( }; }, renderCall(args, _options, theme): Text { + const commandPreview = truncateToWidth(replaceTabs(args.command), 100); return new Text( - `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", args.command)}`, + `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", commandPreview)}`, 0, 0, ); @@ -227,13 +349,13 @@ export function createRunExperimentTool( renderResult(result, options, theme): Text { if (isProgressDetails(result.details)) { const header = theme.fg("warning", `Running ${result.details.elapsed}...`); - const preview = result.content.find(part => part.type === "text")?.text ?? ""; + const preview = replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""); return new Text(preview ? `${header}\n${theme.fg("dim", preview)}` : header, 0, 0); } const details = result.details; if (!details || !isRunDetails(details)) { - return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + return new Text(replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""), 0, 0); } const statusText = renderStatus(details, theme); @@ -241,54 +363,60 @@ export function createRunExperimentTool( return new Text(statusText, 0, 0); } - const preview = options.expanded ? details.tailOutput : details.tailOutput.split("\n").slice(-5).join("\n"); + const preview = replaceTabs( + options.expanded ? details.tailOutput : details.tailOutput.split("\n").slice(-5).join("\n"), + ); const suffix = options.expanded && details.truncation && details.fullOutputPath - ? `\n${theme.fg("warning", `Full output: ${details.fullOutputPath}`)}` + ? `\n${theme.fg("warning", `Full output: ${shortenPath(details.fullOutputPath)}`)}` : ""; return new Text(preview ? `${statusText}\n${theme.fg("dim", preview)}${suffix}` : statusText, 0, 0); }, }; } -interface ProgressSnapshot { - elapsed: string; - fullOutputPath?: string; - tailOutput: string; - truncation?: RunExperimentProgressDetails["truncation"]; -} - async function executeProcess(options: { - command: string; + command: string[]; cwd: string; + logPath: string; timeoutMs: number; signal?: AbortSignal; - onProgress(details: ProgressSnapshot): void; + onProgress?(details: ProgressSnapshot): void; }): Promise { const { promise, resolve, reject } = Promise.withResolvers(); - const child = childProcess.spawn("bash", ["-lc", options.command], { + const child = childProcess.spawn(options.command[0] ?? "bash", options.command.slice(1), { cwd: options.cwd, detached: true, stdio: ["ignore", "pipe", "pipe"], }); - const getTempFile = createTempFileAllocator(); - const chunks: Buffer[] = []; + const tailChunks: Buffer[] = []; let chunksBytes = 0; - let totalBytes = 0; let killedByTimeout = false; let resolved = false; - let fullOutputPath: string | undefined; - let writeStream: fs.WriteStream | undefined; + let writeStream: fs.WriteStream | undefined = fs.createWriteStream(options.logPath); + let forceKillTimeout: NodeJS.Timeout | undefined; + + const closeWriteStream = (): Promise => { + if (!writeStream) return Promise.resolve(); + const stream = writeStream; + writeStream = undefined; + return new Promise((resolveClose, rejectClose) => { + stream.end((error?: Error | null) => { + if (error) { + rejectClose(error); + return; + } + resolveClose(); + }); + }); + }; const cleanup = (): void => { if (progressTimer) clearInterval(progressTimer); if (timeoutHandle) clearTimeout(timeoutHandle); + if (forceKillTimeout) clearTimeout(forceKillTimeout); options.signal?.removeEventListener("abort", abortHandler); - if (writeStream) { - writeStream.end(); - writeStream = undefined; - } }; const finish = (callback: () => void): void => { @@ -299,50 +427,54 @@ async function executeProcess(options: { }; const appendChunk = (data: Buffer): void => { - totalBytes += data.length; - if (!fullOutputPath && totalBytes > DEFAULT_MAX_BYTES) { - fullOutputPath = getTempFile(); - writeStream = fs.createWriteStream(fullOutputPath); - for (const chunk of chunks) { - writeStream.write(chunk); - } - } writeStream?.write(data); - chunks.push(data); + tailChunks.push(data); chunksBytes += data.length; - while (chunksBytes > DEFAULT_MAX_BYTES * 2 && chunks.length > 1) { - const removed = chunks.shift(); + while (chunksBytes > DEFAULT_MAX_BYTES * 2 && tailChunks.length > 1) { + const removed = tailChunks.shift(); if (removed) chunksBytes -= removed.length; } }; const snapshot = (): ProgressSnapshot => { - const tail = truncateTail(Buffer.concat(chunks).toString("utf8"), { + const tail = truncateTail(Buffer.concat(tailChunks).toString("utf8"), { maxBytes: DEFAULT_MAX_BYTES, maxLines: DEFAULT_MAX_LINES, }); return { elapsed: formatElapsed(Date.now() - startedAt), - fullOutputPath, + runDirectory: path.dirname(options.logPath), + fullOutputPath: options.logPath, tailOutput: tail.content, truncation: tail.truncated ? tail : undefined, }; }; + const killTreeWithEscalation = (): void => { + if (!child.pid) return; + killTree(child.pid); + forceKillTimeout = setTimeout(() => { + if (child.pid) killTree(child.pid, "SIGKILL"); + }, 1_000); + forceKillTimeout.unref?.(); + }; + const startedAt = Date.now(); - const progressTimer = setInterval(() => { - options.onProgress(snapshot()); - }, 1000); + const progressTimer = options.onProgress + ? setInterval(() => { + options.onProgress?.(snapshot()); + }, 1000) + : undefined; const timeoutHandle = options.timeoutMs > 0 ? setTimeout(() => { killedByTimeout = true; - if (child.pid) killTree(child.pid); + killTreeWithEscalation(); }, options.timeoutMs) : undefined; const abortHandler = (): void => { - if (child.pid) killTree(child.pid); + killTreeWithEscalation(); }; if (options.signal?.aborted) { abortHandler(); @@ -357,50 +489,59 @@ async function executeProcess(options: { appendChunk(data); }); child.on("error", error => { - finish(() => reject(error)); + void closeWriteStream().finally(() => { + finish(() => reject(error)); + }); }); - child.on("close", code => { - if (options.signal?.aborted) { - finish(() => reject(new Error("aborted"))); - return; + child.on("close", async code => { + try { + await closeWriteStream(); + if (options.signal?.aborted) { + finish(() => reject(new Error("aborted"))); + return; + } + const output = await fs.promises.readFile(options.logPath, "utf8"); + finish(() => + resolve({ + exitCode: code, + killed: killedByTimeout, + logPath: options.logPath, + output, + }), + ); + } catch (error) { + finish(() => reject(error)); } - const output = Buffer.concat(chunks).toString("utf8"); - finish(() => - resolve({ - actualTotalBytes: totalBytes, - exitCode: code, - killed: killedByTimeout, - output, - tempFilePath: fullOutputPath, - }), - ); }); return promise; } -function runChecks(options: { +async function runChecks(options: { cwd: string; pathToChecks: string; + logPath: string; timeoutMs: number; signal?: AbortSignal; - // signal currently unused because spawnSync does not support AbortSignal directly. -}): ChecksExecutionResult { - const result = childProcess.spawnSync("bash", [options.pathToChecks], { +}): Promise { + const result = await executeProcess({ + command: ["bash", options.pathToChecks], cwd: options.cwd, - timeout: options.timeoutMs, - encoding: "utf8", - maxBuffer: DEFAULT_MAX_BYTES, + logPath: options.logPath, + timeoutMs: options.timeoutMs, + signal: options.signal, }); return { - code: result.status, - killed: result.signal === "SIGTERM" || result.signal === "SIGKILL" || Boolean(result.error), - output: `${result.stdout ?? ""}${result.stderr ?? ""}`.trim(), + code: result.exitCode, + killed: result.killed, + logPath: result.logPath, + output: result.output.trim(), }; } function buildRunText(details: RunDetails, outputPreview: string, bestMetric: number | null): string { const lines: string[] = []; + lines.push(`Run directory: ${details.runDirectory}`); if (details.timedOut) { lines.push(`TIMEOUT after ${details.durationSeconds.toFixed(1)}s`); } else if (details.exitCode !== 0) { @@ -420,13 +561,16 @@ function buildRunText(details: RunDetails, outputPreview: string, bestMetric: nu } if (details.parsedPrimary !== null) { lines.push(`Parsed ${details.metricName}: ${details.parsedPrimary}`); + lines.push(`Next log_experiment metric: ${details.parsedPrimary}`); } if (details.parsedMetrics) { - const secondary = Object.entries(details.parsedMetrics) + const secondaryEntries = Object.entries(details.parsedMetrics) .filter(([name]) => name !== details.metricName) - .map(([name, value]) => `${name}=${value}`); + .map(([name, value]) => [name, value] as const); + const secondary = secondaryEntries.map(([name, value]) => `${name}=${value}`); if (secondary.length > 0) { lines.push(`Parsed metrics: ${secondary.join(", ")}`); + lines.push(`Next log_experiment metrics: ${JSON.stringify(Object.fromEntries(secondaryEntries))}`); } } if (details.parsedAsi) { @@ -440,6 +584,9 @@ function buildRunText(details: RunDetails, outputPreview: string, bestMetric: nu `Output truncated (${formatBytes(EXPERIMENT_MAX_BYTES)} limit). Full output: ${details.fullOutputPath}`, ); } + if (details.checksLogPath) { + lines.push(`Checks log: ${details.checksLogPath}`); + } if (details.checksPass === false && details.checksOutput.length > 0) { lines.push(""); lines.push("Checks output:"); @@ -477,3 +624,13 @@ function isProgressDetails(value: unknown): value is RunExperimentProgressDetail if (typeof value !== "object" || value === null) return false; return "phase" in value && value.phase === "running"; } + +function collectLoggedRunNumbers(results: Array<{ runNumber: number | null }>): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} diff --git a/packages/coding-agent/src/autoresearch/types.ts b/packages/coding-agent/src/autoresearch/types.ts index 52f7b89ec..27ec82d02 100644 --- a/packages/coding-agent/src/autoresearch/types.ts +++ b/packages/coding-agent/src/autoresearch/types.ts @@ -21,7 +21,23 @@ export interface MetricDef { unit: string; } +export interface AutoresearchBenchmarkContract { + command: string | null; + primaryMetric: string | null; + metricUnit: string; + direction: MetricDirection | null; + secondaryMetrics: string[]; +} + +export interface AutoresearchContract { + benchmark: AutoresearchBenchmarkContract; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; +} + export interface ExperimentResult { + runNumber: number | null; commit: string; metric: number; metrics: NumericMetricMap; @@ -44,6 +60,11 @@ export interface ExperimentState { currentSegment: number; maxExperiments: number | null; confidence: number | null; + benchmarkCommand: string | null; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; + segmentFingerprint: string | null; } export interface RunExperimentProgressDetails { @@ -51,9 +72,14 @@ export interface RunExperimentProgressDetails { elapsed: string; truncation?: TruncationResult; fullOutputPath?: string; + runDirectory?: string; } export interface RunDetails { + runNumber: number; + runDirectory: string; + benchmarkLogPath: string; + checksLogPath?: string; command: string; exitCode: number | null; durationSeconds: number; @@ -86,20 +112,36 @@ export interface ChecksResult { duration: number; } +export interface PendingRunSummary { + checksDurationSeconds: number | null; + checksPass: boolean | null; + checksTimedOut: boolean; + command: string; + durationSeconds: number | null; + parsedAsi: ASIData | null; + parsedMetrics: NumericMetricMap | null; + parsedPrimary: number | null; + passed: boolean; + runDirectory: string; + runNumber: number; +} + export interface RunningExperiment { startedAt: number; command: string; + runDirectory: string; + runNumber: number; } export interface AutoresearchRuntime { autoresearchMode: boolean; dashboardExpanded: boolean; - lastAutoResumeTime: number; - experimentsThisSession: number; - autoResumeTurns: number; lastRunChecks: ChecksResult | null; lastRunDuration: number | null; lastRunAsi: ASIData | null; + lastRunArtifactDir: string | null; + lastRunNumber: number | null; + lastRunSummary: PendingRunSummary | null; runningExperiment: RunningExperiment | null; state: ExperimentState; goal: string | null; @@ -116,6 +158,12 @@ export interface AutoresearchJsonConfigEntry { metricName?: string; metricUnit?: string; bestDirection?: MetricDirection; + benchmarkCommand?: string; + secondaryMetrics?: string[]; + scopePaths?: string[]; + offLimits?: string[]; + constraints?: string[]; + segmentFingerprint?: string; } export interface AutoresearchJsonRunEntry { @@ -143,6 +191,7 @@ export interface AutoresearchControlEntryData { export interface ReconstructedControlState { autoresearchMode: boolean; goal: string | null; + lastMode: AutoresearchControlEntryData["mode"] | null; } export interface RuntimeStore { diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 6f556a6dd..c61fd3d9b 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -1054,7 +1054,13 @@ export interface ExtensionAPI { // Actions // ========================================================================= - /** Send a custom message to the session. */ + /** + * Send a custom message to the session. + * + * `deliverAs: "nextTurn"` keeps the message hidden from the editable pending-message UI. + * If `triggerTurn` is also true while the current turn is still unwinding, the session schedules + * an internal continuation that consumes the message on the next turn. + */ sendMessage( message: Pick, "customType" | "content" | "display" | "details" | "attribution">, options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" }, @@ -1230,6 +1236,11 @@ type HandlerFn = (...args: unknown[]) => Promise; export type SendMessageHandler = ( message: Pick, "customType" | "content" | "display" | "details" | "attribution">, + /** + * `deliverAs: "nextTurn"` queues hidden custom context for the next turn. + * When paired with `triggerTurn: true` during prompt teardown, the session schedules + * an internal continuation without surfacing the message in the editable pending queue. + */ options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" }, ) => void; diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 108c777c7..ffe123444 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -364,6 +364,7 @@ export class AgentSession { #followUpMessages: string[] = []; /** Messages queued to be included with the next user prompt as context ("asides"). */ #pendingNextTurnMessages: CustomMessage[] = []; + #scheduledHiddenNextTurnGeneration: number | undefined = undefined; #planModeState: PlanModeState | undefined; #planReferenceSent = false; #planReferencePath = "local://PLAN.md"; @@ -2567,6 +2568,74 @@ export class AgentSession { }); } + #queueHiddenNextTurnMessage(message: CustomMessage, triggerTurn: boolean): void { + this.#pendingNextTurnMessages.push(message); + if (!triggerTurn) return; + const generation = this.#promptGeneration; + if (this.#scheduledHiddenNextTurnGeneration === generation) { + return; + } + this.#scheduledHiddenNextTurnGeneration = generation; + this.#schedulePostPromptTask( + async () => { + if (this.#scheduledHiddenNextTurnGeneration === generation) { + this.#scheduledHiddenNextTurnGeneration = undefined; + } + if (this.#pendingNextTurnMessages.length === 0) { + return; + } + try { + await this.#promptQueuedHiddenNextTurnMessages(); + } catch { + // Leave the hidden next-turn messages queued for the next explicit prompt. + } + }, + { + generation, + onSkip: () => { + if (this.#scheduledHiddenNextTurnGeneration === generation) { + this.#scheduledHiddenNextTurnGeneration = undefined; + } + }, + }, + ); + } + + async #promptQueuedHiddenNextTurnMessages(): Promise { + if (this.#pendingNextTurnMessages.length === 0) { + return; + } + + const queuedMessages = [...this.#pendingNextTurnMessages]; + this.#pendingNextTurnMessages = []; + const message = queuedMessages[queuedMessages.length - 1]; + if (!message) { + return; + } + + const prependMessages = queuedMessages.slice(0, -1); + const textContent = this.#getCustomMessageTextContent(message); + try { + await this.#promptWithMessage(message, textContent, { + prependMessages, + skipPostPromptRecoveryWait: true, + }); + } catch (error) { + this.#pendingNextTurnMessages = [...queuedMessages, ...this.#pendingNextTurnMessages]; + throw error; + } + } + + #getCustomMessageTextContent(message: Pick): string { + if (typeof message.content === "string") { + return message.content; + } + return message.content + .filter((content): content is TextContent => content.type === "text") + .map(content => content.text) + .join(""); + } + /** * Throw an error if the text is an extension command. */ @@ -2607,7 +2676,7 @@ export class AgentSession { }; if (this.isStreaming) { if (options?.deliverAs === "nextTurn") { - this.#pendingNextTurnMessages.push(appMessage); + this.#queueHiddenNextTurnMessage(appMessage, options?.triggerTurn ?? false); return; } @@ -2619,6 +2688,22 @@ export class AgentSession { return; } + if (options?.deliverAs === "nextTurn") { + if (options?.triggerTurn) { + await this.agent.prompt(appMessage); + return; + } + this.agent.appendMessage(appMessage); + this.sessionManager.appendCustomMessageEntry( + message.customType, + message.content, + message.display, + message.details, + message.attribution ?? "agent", + ); + return; + } + if (options?.triggerTurn) { await this.agent.prompt(appMessage); return; @@ -2686,9 +2771,9 @@ export class AgentSession { return { steering, followUp }; } - /** Number of pending messages (includes both steering and follow-up) */ + /** Number of pending messages (includes steering, follow-up, and next-turn messages) */ get queuedMessageCount(): number { - return this.#steeringMessages.length + this.#followUpMessages.length; + return this.#steeringMessages.length + this.#followUpMessages.length + this.#pendingNextTurnMessages.length; } /** Get pending messages (read-only) */ @@ -2830,6 +2915,7 @@ export class AgentSession { async abort(): Promise { this.abortRetry(); this.#promptGeneration++; + this.#scheduledHiddenNextTurnGeneration = undefined; this.#resolveTtsrResume(); this.#cancelPostPromptTasks(); this.agent.abort(); @@ -2879,6 +2965,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; this.sessionManager.appendThinkingLevelChange(this.thinkingLevel); this.sessionManager.appendServiceTierChange(this.serviceTier ?? null); @@ -3612,6 +3699,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; this.#todoReminderCount = 0; // Inject the handoff document as a custom message @@ -4961,6 +5049,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; // Flush pending writes before switching await this.sessionManager.flush(); @@ -5060,6 +5149,7 @@ export class AgentSession { // Clear pending messages (bound to old session state) this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; // Flush pending writes before branching await this.sessionManager.flush(); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 88408f1d6..42ccc162a 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -2,11 +2,11 @@ * Tests for AgentSession concurrent prompt guard. */ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { Agent, AgentBusyError, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; import { type AssistantMessage, getBundledModel, type ToolCall } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; @@ -62,6 +62,7 @@ describe("AgentSession concurrent prompt guard", () => { if (tempDir && fs.existsSync(tempDir)) { fs.rmSync(tempDir, { recursive: true }); } + vi.restoreAllMocks(); }); async function createSession() { @@ -163,6 +164,76 @@ describe("AgentSession concurrent prompt guard", () => { await firstPrompt.catch(() => {}); }); + it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let firstStream: MockAssistantStream | undefined; + const callMessages: AgentMessage[][] = []; + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { + model, + systemPrompt: "Test", + tools: [], + }, + streamFn: (_model, context) => { + callMessages.push([...context.messages]); + const stream = new MockAssistantStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: createAssistantMessage("") }); + if (callMessages.length > 1) { + stream.push({ type: "done", reason: "stop", message: createAssistantMessage("Resumed") }); + return; + } + }); + firstStream = stream; + return stream; + }, + }); + + const sessionManager = SessionManager.inMemory(); + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + + session = new AgentSession({ + agent, + sessionManager, + settings, + modelRegistry, + }); + + const firstPrompt = session.prompt("First message"); + await Bun.sleep(10); + + await session.sendCustomMessage( + { + customType: "autoresearch-resume", + content: "Hidden stop reaction", + display: false, + attribution: "agent", + }, + { deliverAs: "nextTurn", triggerTurn: true }, + ); + + expect(session.queuedMessageCount).toBe(0); + expect(session.getQueuedMessages()).toEqual({ steering: [], followUp: [] }); + + firstStream?.push({ type: "done", reason: "stop", message: createAssistantMessage("Done") }); + await firstPrompt; + await session.waitForIdle(); + + expect(callMessages).toHaveLength(2); + expect( + callMessages[1]?.some( + message => + message.role === "custom" && "customType" in message && message.customType === "autoresearch-resume", + ), + ).toBe(true); + }); + it("should allow prompt() after previous completes", async () => { // Create session with a stream that completes immediately const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; diff --git a/packages/coding-agent/test/autoresearch-state.test.ts b/packages/coding-agent/test/autoresearch-state.test.ts index 285cd70da..a5d20eb17 100644 --- a/packages/coding-agent/test/autoresearch-state.test.ts +++ b/packages/coding-agent/test/autoresearch-state.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Snowflake } from "@oh-my-pi/pi-utils"; +import { parseAutoresearchContract } from "../src/autoresearch/contract"; import { isAutoresearchShCommand } from "../src/autoresearch/helpers"; import { createAutoresearchExtension } from "../src/autoresearch/index"; import { reconstructStateFromJsonl } from "../src/autoresearch/state"; @@ -14,6 +15,7 @@ import type { RegisteredCommand, SessionStartEvent, SessionSwitchEvent, + ToolCallEvent, } from "../src/extensibility/extensions"; function makeTempDir(): string { @@ -100,6 +102,187 @@ describe("autoresearch state reconstruction", () => { expect(state.results.filter(result => result.segment === 1)).toHaveLength(2); expect(state.secondaryMetrics).toEqual([{ name: "latency_ms", unit: "ms" }]); }); + + it("hydrates configured secondary metrics from config entries before later runs add new ones", () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const jsonlPath = path.join(dir, "autoresearch.jsonl"); + fs.writeFileSync( + jsonlPath, + [ + JSON.stringify({ + type: "config", + name: "Baseline", + metricName: "runtime_ms", + metricUnit: "ms", + bestDirection: "lower", + secondaryMetrics: ["memory_mb", "tokens"], + }), + JSON.stringify({ + commit: "aaaaaaa", + metric: 100, + metrics: { memory_mb: 32 }, + status: "keep", + description: "baseline", + timestamp: 1, + }), + ].join("\n"), + ); + + const reconstructed = reconstructStateFromJsonl(dir); + expect(reconstructed.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + }); + + it("uses the first kept run as baseline and preserves configured secondary metrics before they appear", () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const jsonlPath = path.join(dir, "autoresearch.jsonl"); + fs.writeFileSync( + jsonlPath, + [ + JSON.stringify({ + type: "config", + name: "Baseline after crash", + metricName: "runtime_ms", + metricUnit: "ms", + bestDirection: "lower", + secondaryMetrics: ["memory_mb", "tokens"], + }), + JSON.stringify({ + commit: "aaaaaaa", + metric: 0, + status: "crash", + description: "broken first run", + timestamp: 1, + }), + JSON.stringify({ + commit: "bbbbbbb", + metric: 120, + metrics: { memory_mb: 32 }, + status: "keep", + description: "baseline", + timestamp: 2, + }), + ].join("\n"), + ); + + const reconstructed = reconstructStateFromJsonl(dir); + expect(reconstructed.state.bestMetric).toBe(120); + expect(reconstructed.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + }); + + it("parses benchmark, scope, off-limits, and constraints from autoresearch.md", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower +- secondary metrics: memory_mb, tokens + +## Files in Scope +- src/core +- src/feature.ts + +## Off Limits +- src/generated + +## Constraints +- keep API stable +- no behavior regressions +`); + + expect(contract.benchmark.command).toBe("bash autoresearch.sh"); + expect(contract.benchmark.primaryMetric).toBe("runtime_ms"); + expect(contract.benchmark.metricUnit).toBe("ms"); + expect(contract.benchmark.direction).toBe("lower"); + expect(contract.benchmark.secondaryMetrics).toEqual(["memory_mb", "tokens"]); + expect(contract.scopePaths).toEqual(["src/core", "src/feature.ts"]); + expect(contract.offLimits).toEqual(["src/generated"]); + expect(contract.constraints).toEqual(["keep API stable", "no behavior regressions"]); + }); + + it("parses nested secondary metric bullets from autoresearch.md", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower +- secondary metrics: + - memory_mb + - rss_mb + +## Files in Scope +- src +`); + + expect(contract.benchmark.secondaryMetrics).toEqual(["memory_mb", "rss_mb"]); + }); + + it("allows empty optional sections while preserving an empty off-limits list", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: +- direction: higher + +## Files in Scope +- . + +## Off Limits + +## Constraints +`); + + expect(contract.benchmark.metricUnit).toBe(""); + expect(contract.benchmark.direction).toBe("higher"); + expect(contract.scopePaths).toEqual(["."]); + expect(contract.offLimits).toEqual([]); + expect(contract.constraints).toEqual([]); + }); + + it("preserves free-form constraint text without path normalization", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower + +## Files in Scope +- src/ + +## Off Limits +- generated/ + +## Constraints +- keep docs/ wording exactly as written +- do not rewrite ./README.md examples +`); + + expect(contract.scopePaths).toEqual(["src"]); + expect(contract.offLimits).toEqual(["generated"]); + expect(contract.constraints).toEqual([ + "keep docs/ wording exactly as written", + "do not rewrite ./README.md examples", + ]); + }); }); describe("autoresearch command guard", () => { @@ -127,7 +310,7 @@ interface AutoresearchCommandHarness { function createAutoresearchCommandHarness( cwd: string, - inputResult: string | undefined, + inputResult: string | string[] | undefined, execImpl?: (command: string, args: string[]) => Promise<{ code: number; stderr: string; stdout: string }>, ): AutoresearchCommandHarness { const execCalls: Array<{ args: string[]; command: string }> = []; @@ -135,6 +318,7 @@ function createAutoresearchCommandHarness( const inputCalls: Array<{ title: string; placeholder: string | undefined }> = []; const notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }> = []; let command: RegisteredCommand | undefined; + const inputQueue = typeof inputResult === "string" || inputResult === undefined ? [inputResult] : [...inputResult]; const api = { appendEntry(_customType: string, _data?: unknown): void {}, @@ -178,6 +362,7 @@ function createAutoresearchCommandHarness( newSession: async () => ({ cancelled: false }), reload: async () => {}, sessionManager: { + getBranch: () => [], getEntries: () => [], getSessionId: () => "session-1", }, @@ -188,7 +373,7 @@ function createAutoresearchCommandHarness( custom: async () => undefined, input: async (title: string, placeholder?: string) => { inputCalls.push({ title, placeholder }); - return inputResult; + return inputQueue.shift(); }, notify(message: string, type?: "info" | "warning" | "error"): void { notifications.push({ message, type }); @@ -211,17 +396,23 @@ function createAutoresearchCommandHarness( interface AutoresearchLifecycleHarness { sessionStartHandler: ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) | undefined; sessionSwitchHandler: ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined; + agentEndHandler: ((event: unknown, ctx: ExtensionContext) => Promise | void) | undefined; + toolCallHandler: ((event: ToolCallEvent, ctx: ExtensionContext) => Promise | unknown) | undefined; ctx: ExtensionContext; setActiveToolsCalls: string[][]; + sentMessages: Array<{ message: unknown; options: unknown }>; } function createAutoresearchLifecycleHarness(options: { activeTools: string[]; + branchEntries?: Array<{ type: "custom"; customType: string; data?: unknown }>; controlEntries?: Array<{ type: "custom"; customType: string; data?: unknown }>; + cwd?: string; }): AutoresearchLifecycleHarness { const handlers = new Map Promise | void>(); const activeTools = [...options.activeTools]; const setActiveToolsCalls: string[][] = []; + const sentMessages: Array<{ message: unknown; options: unknown }> = []; const api = { appendEntry(_customType: string, _data?: unknown): void {}, @@ -234,6 +425,9 @@ function createAutoresearchLifecycleHarness(options: { getActiveTools(): string[] { return [...activeTools]; }, + sendMessage(message: unknown, options?: unknown): void { + sentMessages.push({ message, options }); + }, async setActiveTools(toolNames: string[]): Promise { setActiveToolsCalls.push([...toolNames]); activeTools.splice(0, activeTools.length, ...toolNames); @@ -245,7 +439,7 @@ function createAutoresearchLifecycleHarness(options: { const ctx = { abort(): void {}, compact: async () => {}, - cwd: makeTempDir(), + cwd: options.cwd ?? makeTempDir(), getContextUsage: () => undefined, hasUI: false, hasPendingMessages: () => false, @@ -253,6 +447,7 @@ function createAutoresearchLifecycleHarness(options: { model: undefined, modelRegistry: {}, sessionManager: { + getBranch: () => options.branchEntries ?? options.controlEntries ?? [], getEntries: () => options.controlEntries ?? [], getSessionId: () => "session-1", }, @@ -286,8 +481,15 @@ function createAutoresearchLifecycleHarness(options: { sessionSwitchHandler: handlers.get("session_switch") as | ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined, + agentEndHandler: handlers.get("agent_end") as + | ((event: unknown, ctx: ExtensionContext) => Promise | void) + | undefined, + toolCallHandler: handlers.get("tool_call") as + | ((event: ToolCallEvent, ctx: ExtensionContext) => Promise | unknown) + | undefined, ctx, setActiveToolsCalls, + sentMessages, }; } @@ -307,7 +509,16 @@ describe("autoresearch command startup", () => { const branches = new Set(); const harness = createAutoresearchCommandHarness( dir, - "reduce edit benchmark runtime variance", + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh --quick", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch, packages/coding-agent/test", + "packages/coding-agent/src/generated", + "preserve output format", + ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; @@ -332,13 +543,27 @@ describe("autoresearch command startup", () => { expect(harness.inputCalls).toEqual([ { title: "Autoresearch Intent", placeholder: "what should autoresearch improve?" }, + { title: "Benchmark Command", placeholder: "bash autoresearch.sh" }, + { title: "Primary Metric Name", placeholder: "runtime_ms" }, + { title: "Metric Unit", placeholder: "ms" }, + { title: "Metric Direction", placeholder: "lower" }, + { title: "Files in Scope", placeholder: "packages/coding-agent/src/autoresearch" }, + { title: "Off Limits", placeholder: "" }, + { title: "Constraints", placeholder: "" }, ]); expect(harness.sentMessages).toHaveLength(1); expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); expect(harness.sentMessages[0]).toContain("reduce edit benchmark runtime variance"); + expect(harness.sentMessages[0]).toContain("benchmark command: `bash autoresearch.sh --quick`"); + expect(harness.sentMessages[0]).toContain("primary metric: `runtime_ms`"); + expect(harness.sentMessages[0]).toContain("metric unit: `ms`"); + expect(harness.sentMessages[0]).toContain("direction: `lower`"); + expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/autoresearch`"); + expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/generated`"); + expect(harness.sentMessages[0]).toContain("preserve output format"); expect(harness.sentMessages[0]).toContain("Created and checked out dedicated git branch"); expect(harness.sentMessages[0]).toContain("Explain briefly what autoresearch will do in this repository"); - expect(harness.sentMessages[0]).toContain("Files in Scope"); + expect(harness.sentMessages[0]).toContain("- files in scope:"); expect(harness.notifications).toEqual([]); const checkoutCall = harness.execCalls.find(call => call.command === "git" && call.args[0] === "checkout"); expect(checkoutCall?.args[2]).toMatch(/^autoresearch\/reduce-edit-benchmark-runtime-variance-\d{8}$/); @@ -352,6 +577,7 @@ describe("autoresearch command startup", () => { const harness = createAutoresearchCommandHarness(dir, "ignored", async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; if (args[0] === "branch" && args[1] === "--show-current") { return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; } @@ -378,6 +604,108 @@ describe("autoresearch command startup", () => { ]); }); + it("includes explicit resume context when the user resumes with additional instructions", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const autoresearchMdPath = path.join(dir, "autoresearch.md"); + fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); + await Bun.write(path.join(dir, ".autoresearch", "runs", "0001", "run.json"), "{}"); + const harness = createAutoresearchCommandHarness(dir, undefined, async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }); + + await harness.command.handler("focus on memory regressions next", harness.ctx); + + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Additional context from the user:"); + expect(harness.sentMessages[0]).toContain("focus on memory regressions next"); + expect(harness.sentMessages[0]).toContain(`@${autoresearchMdPath}`); + }); + + it("treats an explicit new intent as a fresh setup when only stale notes remain", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync(path.join(dir, "autoresearch.md"), "# Autoresearch\n\nOld notes\n"); + let currentBranch = "main"; + const branches = new Set(); + const harness = createAutoresearchCommandHarness( + dir, + [ + "focus on memory regressions next", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: `${currentBranch}\n` }; + } + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; + if (args[0] === "show-ref") { + const branchName = args[args.length - 1]?.replace("refs/heads/", "") ?? ""; + return { code: branches.has(branchName) ? 0 : 1, stderr: "", stdout: "" }; + } + if (args[0] === "checkout" && args[1] === "-b") { + currentBranch = args[2] ?? currentBranch; + branches.add(currentBranch); + return { code: 0, stderr: "", stdout: "" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await harness.command.handler("focus on memory regressions next", harness.ctx); + + expect(harness.inputCalls[0]).toEqual({ + title: "Autoresearch Intent", + placeholder: "focus on memory regressions next", + }); + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); + expect(harness.sentMessages[0]).not.toContain("Resume autoresearch from the attached notes."); + }); + + it("refuses to resume on an autoresearch branch when non-local files are dirty", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const autoresearchMdPath = path.join(dir, "autoresearch.md"); + fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); + const harness = createAutoresearchCommandHarness(dir, "ignored", async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: " M packages/coding-agent/src/sdk.ts\0" }; + } + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([ + { + message: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", + type: "error", + }, + ]); + }); + it("does not start autoresearch when the intent dialog returns blank input", async () => { const dir = makeTempDir(); tempDirs.push(dir); @@ -389,12 +717,34 @@ describe("autoresearch command startup", () => { expect(harness.notifications).toEqual([{ message: "Autoresearch intent is required", type: "info" }]); }); + it("rejects non-canonical benchmark commands during setup", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const harness = createAutoresearchCommandHarness(dir, ["speed things up", "pnpm test"]); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([ + { message: "Benchmark command must invoke `autoresearch.sh` directly", type: "info" }, + ]); + }); + it("refuses to start when non-autoresearch files are dirty on a non-autoresearch branch", async () => { const dir = makeTempDir(); tempDirs.push(dir); const harness = createAutoresearchCommandHarness( dir, - "reduce edit benchmark runtime variance", + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; @@ -414,11 +764,299 @@ describe("autoresearch command startup", () => { expect(harness.notifications).toEqual([ { message: - "Autoresearch needs a clean git worktree before it can create an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", type: "error", }, ]); }); + + it("ignores autoresearch local state but still blocks dirty control files before creating a branch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + + const localStateHarness = createAutoresearchCommandHarness( + dir, + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "main\n" }; + } + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: "?? autoresearch.jsonl\n?? .autoresearch/runs/0001/run.json\n" }; + } + if (args[0] === "show-ref") return { code: 1, stderr: "", stdout: "" }; + if (args[0] === "checkout" && args[1] === "-b") return { code: 0, stderr: "", stdout: "" }; + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await localStateHarness.command.handler("", localStateHarness.ctx); + + expect(localStateHarness.sentMessages).toHaveLength(1); + expect(localStateHarness.notifications).toEqual([]); + + const dirtyControlHarness = createAutoresearchCommandHarness( + dir, + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "main\n" }; + } + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: " M autoresearch.md\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await dirtyControlHarness.command.handler("", dirtyControlHarness.ctx); + + expect(dirtyControlHarness.sentMessages).toEqual([]); + expect(dirtyControlHarness.notifications).toEqual([ + { + message: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: autoresearch.md", + type: "error", + }, + ]); + }); +}); + +describe("autoresearch tool-call guard", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("blocks out-of-scope edits but allows autoresearch control files", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ + type: "config", + metricName: "runtime_ms", + metricUnit: "ms", + scopePaths: ["src"], + offLimits: ["src/generated"], + })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blockedScope = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-1", + toolName: "write", + input: { path: "README.md", content: "nope" }, + }, + harness.ctx, + ); + expect(blockedScope).toEqual({ + block: true, + reason: expect.stringContaining("outside Files in Scope"), + }); + + const blockedLocalState = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-2", + toolName: "write", + input: { path: "autoresearch.jsonl", content: "[]" }, + }, + harness.ctx, + ); + expect(blockedLocalState).toEqual({ + block: true, + reason: expect.stringContaining("local state files"), + }); + + const allowedControl = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-3", + toolName: "write", + input: { path: "autoresearch.program.md", content: "# Strategy" }, + }, + harness.ctx, + ); + expect(allowedControl).toBeUndefined(); + }); + + it("requires ast_edit to declare an explicit path during autoresearch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-ast", + toolName: "ast_edit", + input: { ops: [{ pat: "a", out: "b" }] }, + }, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("explicit target path"), + }); + }); + + it("blocks mutating bash commands during autoresearch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-bash", + toolName: "bash", + input: { command: "rm -rf src/generated" }, + } as ToolCallEvent, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("read-only shell inspection"), + }); + }); + + it("blocks symlink escapes that point outside the working tree", async () => { + const dir = makeTempDir(); + const outsideDir = makeTempDir(); + tempDirs.push(dir, outsideDir); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.symlinkSync(outsideDir, path.join(dir, "src", "linked-outside"), "dir"); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-symlink", + toolName: "write", + input: { path: "src/linked-outside/escape.ts", content: "export const value = 1;\n" }, + }, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("outside the working tree"), + }); + }); +}); + +describe("autoresearch auto-resume", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("includes the pending-run reminder after rehydrate when agent_end schedules an auto-resume", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", metricName: "runtime_ms", scopePaths: ["src"] })}\n`, + ); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["init_experiment", "run_experiment", "log_experiment"], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + await harness.agentEndHandler?.({}, harness.ctx); + + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]?.message).toMatchObject({ + customType: "autoresearch-resume", + content: expect.stringContaining("finish the pending `log_experiment` step"), + }); + expect(harness.sentMessages[0]?.options).toMatchObject({ + deliverAs: "nextTurn", + triggerTurn: true, + }); + }); }); describe("autoresearch lifecycle tool activation", () => { @@ -449,6 +1087,19 @@ describe("autoresearch lifecycle tool activation", () => { expect(harness.setActiveToolsCalls).toEqual([["read"]]); }); + + it("rehydrates control state from the active branch only", async () => { + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["read"], + branchEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "off" } }], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "speed" } }], + }); + + if (!harness.sessionStartHandler) throw new Error("Expected session_start handler"); + await harness.sessionStartHandler({ type: "session_start" }, harness.ctx); + + expect(harness.setActiveToolsCalls).toEqual([]); + }); }); describe("autoresearch ASI requirements", () => { diff --git a/packages/coding-agent/test/autoresearch-tools.test.ts b/packages/coding-agent/test/autoresearch-tools.test.ts new file mode 100644 index 000000000..068c16452 --- /dev/null +++ b/packages/coding-agent/test/autoresearch-tools.test.ts @@ -0,0 +1,1231 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { $ } from "bun"; +import { + buildAutoresearchSegmentFingerprint, + loadAutoresearchScriptSnapshot, + readAutoresearchContract, +} from "../src/autoresearch/contract"; +import { readPendingRunSummary } from "../src/autoresearch/helpers"; +import { createSessionRuntime } from "../src/autoresearch/state"; +import { createInitExperimentTool } from "../src/autoresearch/tools/init-experiment"; +import { createLogExperimentTool } from "../src/autoresearch/tools/log-experiment"; +import { createRunExperimentTool } from "../src/autoresearch/tools/run-experiment"; +import type { RunDetails } from "../src/autoresearch/types"; +import type { ExtensionAPI, ExtensionContext } from "../src/extensibility/extensions"; + +function makeTempDir(): string { + const dir = path.join(os.tmpdir(), `pi-autoresearch-tools-${Snowflake.next()}`); + fs.mkdirSync(dir, { recursive: true }); + return dir; +} + +function writeAutoresearchWorkspace( + dir: string, + options?: { + checksScript?: string; + contract?: string; + benchmarkScript?: string; + }, +): void { + fs.writeFileSync( + path.join(dir, "autoresearch.md"), + options?.contract ?? + [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + ); + fs.writeFileSync( + path.join(dir, "autoresearch.sh"), + options?.benchmarkScript ?? + [ + "#!/usr/bin/env bash", + "set -euo pipefail", + "echo METRIC runtime_ms=10", + "echo METRIC memory_mb=32", + 'echo ASI hypothesis="baseline"', + ].join("\n"), + ); + fs.chmodSync(path.join(dir, "autoresearch.sh"), 0o755); + if (options?.checksScript) { + fs.writeFileSync(path.join(dir, "autoresearch.checks.sh"), options.checksScript); + fs.chmodSync(path.join(dir, "autoresearch.checks.sh"), 0o755); + } +} + +function createFingerprint(workDir: string): string { + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + if (contractResult.errors.length > 0 || scriptSnapshot.errors.length > 0) { + throw new Error(`Workspace setup invalid: ${[...contractResult.errors, ...scriptSnapshot.errors].join(" ")}`); + } + return buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); +} + +function createDashboardStub() { + return { + clear(): void {}, + requestRender(): void {}, + showOverlay: async (): Promise => {}, + updateWidget(): void {}, + }; +} + +function createContext(cwd: string): ExtensionContext { + return { cwd, hasUI: false } as ExtensionContext; +} + +function createGitApi(): ExtensionAPI { + return { + exec: async (command: string, args: string[], options?: { cwd?: string }) => { + const result = Bun.spawnSync([command, ...args], { + cwd: options?.cwd ?? process.cwd(), + stdout: "pipe", + stderr: "pipe", + }); + return { + code: result.exitCode, + stdout: Buffer.from(result.stdout).toString("utf8"), + stderr: Buffer.from(result.stderr).toString("utf8"), + }; + }, + } as unknown as ExtensionAPI; +} + +function createManagedGitApi(options?: { activeTools?: string[] }) { + const activeTools = [...(options?.activeTools ?? ["init_experiment", "run_experiment", "log_experiment"])]; + const appendEntries: Array<{ customType: string; data: unknown }> = []; + const setActiveToolsCalls: string[][] = []; + const api = { + appendEntry: (customType: string, data?: unknown) => { + appendEntries.push({ customType, data }); + }, + exec: async (command: string, args: string[], execOptions?: { cwd?: string }) => { + const result = Bun.spawnSync([command, ...args], { + cwd: execOptions?.cwd ?? process.cwd(), + stdout: "pipe", + stderr: "pipe", + }); + return { + code: result.exitCode, + stdout: Buffer.from(result.stdout).toString("utf8"), + stderr: Buffer.from(result.stderr).toString("utf8"), + }; + }, + getActiveTools: () => [...activeTools], + setActiveTools: async (toolNames: string[]) => { + setActiveToolsCalls.push([...toolNames]); + activeTools.splice(0, activeTools.length, ...toolNames); + }, + } as unknown as ExtensionAPI; + return { activeTools, api, appendEntries, setActiveToolsCalls }; +} + +function expectRunDetails(details: unknown): RunDetails { + if (!details || typeof details !== "object" || !("benchmarkLogPath" in details)) { + throw new Error("Expected run details"); + } + return details as RunDetails; +} + +describe("autoresearch tools", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("writes durable benchmark/check artifacts and run metadata", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + checksScript: ["#!/usr/bin/env bash", "set -euo pipefail", "echo checks ok"].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "call-1", + { command: "bash autoresearch.sh", timeout_seconds: 5, checks_timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + const details = expectRunDetails(result.details); + + expect(details.runNumber).toBe(1); + expect(details.parsedPrimary).toBe(10); + expect(details.parsedMetrics).toEqual({ memory_mb: 32, runtime_ms: 10 }); + expect(details.benchmarkLogPath).toBe(path.join(details.runDirectory, "benchmark.log")); + expect(fs.existsSync(details.benchmarkLogPath)).toBe(true); + expect(fs.existsSync(details.checksLogPath ?? "")).toBe(true); + + const runJson = JSON.parse(fs.readFileSync(path.join(details.runDirectory, "run.json"), "utf8")) as { + completedAt?: string; + parsedPrimary?: number; + checks?: { passed?: boolean }; + }; + expect(runJson.completedAt).toEqual(expect.any(String)); + expect(runJson.parsedPrimary).toBe(10); + expect(runJson.checks?.passed).toBe(true); + }); + + it("ignores incomplete run artifacts until the benchmark has actually finished", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + runNumber: 1, + startedAt: new Date().toISOString(), + }), + ); + + const pendingRun = await readPendingRunSummary(dir); + expect(pendingRun).toBeNull(); + }); + + it("persists init_experiment config metadata from autoresearch.md", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "- secondary metrics: memory_mb, tokens", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "init-1", + { + name: "Reduce runtime variance", + metric_name: "runtime_ms", + metric_unit: "ms", + direction: "lower", + benchmark_command: "bash autoresearch.sh", + scope_paths: ["src"], + off_limits: ["src/generated"], + constraints: ["keep behavior stable"], + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Experiment initialized: Reduce runtime variance"), + }); + expect(runtime.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + + const configEntry = JSON.parse(fs.readFileSync(path.join(dir, "autoresearch.jsonl"), "utf8").trim()) as { + benchmarkCommand?: string; + constraints?: string[]; + offLimits?: string[]; + scopePaths?: string[]; + secondaryMetrics?: string[]; + segmentFingerprint?: string; + }; + expect(configEntry.benchmarkCommand).toBe("bash autoresearch.sh"); + expect(configEntry.secondaryMetrics).toEqual(["memory_mb", "tokens"]); + expect(configEntry.scopePaths).toEqual(["src"]); + expect(configEntry.offLimits).toEqual(["src/generated"]); + expect(configEntry.constraints).toEqual(["keep behavior stable"]); + expect(configEntry.segmentFingerprint).toBe(createFingerprint(dir)); + }); + + it("rejects init_experiment when the passed contract no longer matches autoresearch.md", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "init-2", + { + name: "Mismatch", + metric_name: "runtime_ms", + metric_unit: "ms", + direction: "lower", + benchmark_command: "bash autoresearch.sh", + scope_paths: ["src"], + off_limits: ["src/other-generated"], + constraints: ["keep behavior stable"], + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("off_limits do not match autoresearch.md"), + }); + expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false); + }); + + it("refuses to start a new benchmark while a previous run artifact is still unlogged", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ command: "bash autoresearch.sh", exitCode: 0, parsedPrimary: 10, runNumber: 1 }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-1b", + { command: "bash autoresearch.sh", timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("has not been logged yet"), + }); + }); + + it("refuses to run when the current segment fingerprint is stale", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + fs.writeFileSync( + path.join(dir, "autoresearch.sh"), + ["#!/usr/bin/env bash", "set -euo pipefail", "echo METRIC runtime_ms=9"].join("\n"), + ); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-2", + { command: "bash autoresearch.sh", timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Re-run init_experiment"), + }); + }); + + it("times out checks asynchronously and preserves the checks log", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + checksScript: ["#!/usr/bin/env bash", "set -euo pipefail", "sleep 2", "echo done"].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-3", + { command: "bash autoresearch.sh", timeout_seconds: 5, checks_timeout_seconds: 0.1 }, + undefined, + undefined, + createContext(dir), + ); + const details = expectRunDetails(result.details); + + expect(details.checksTimedOut).toBe(true); + expect(fs.existsSync(details.checksLogPath ?? "")).toBe(true); + }); + + it("honors user aborts while the experiment is running", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + benchmarkScript: ["#!/usr/bin/env bash", "set -euo pipefail", "sleep 5", "echo METRIC runtime_ms=10"].join( + "\n", + ), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const controller = new AbortController(); + setTimeout(() => controller.abort(), 100); + + await expect( + tool.execute( + "call-4", + { command: "bash autoresearch.sh", timeout_seconds: 10 }, + controller.signal, + undefined, + createContext(dir), + ), + ).rejects.toThrow("aborted"); + expect(fs.existsSync(path.join(dir, ".autoresearch", "runs", "0001", "run.json"))).toBe(true); + }); + + it("commits only in-scope changes and excludes autoresearch local state", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src/in-scope.ts", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 1;\n"); + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 2;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 3;\n"); + fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n\n- focus on in-scope edits\n"); + fs.writeFileSync(path.join(dir, "autoresearch.jsonl"), '{"type":"run"}\n'); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src/in-scope.ts"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.constraints = ["keep behavior stable"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + const runDirectory = path.join(dir, ".autoresearch", "runs", "0001"); + runtime.lastRunArtifactDir = runDirectory; + runtime.lastRunNumber = 1; + runtime.lastRunDuration = 1.2; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1.2, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory, + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-5", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Improve in scope", + asi: { hypothesis: "inline the hot path" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Logged run #1: keep"), + }); + const committedPaths = await $`git show --name-only --pretty=format: HEAD`.cwd(dir).text(); + expect(committedPaths).toContain("src/in-scope.ts"); + expect(committedPaths).toContain("autoresearch.program.md"); + expect(committedPaths).not.toContain("autoresearch.jsonl"); + expect(committedPaths).not.toContain(".autoresearch"); + + const runJson = JSON.parse(fs.readFileSync(path.join(runDirectory, "run.json"), "utf8")) as { + status?: string; + }; + expect(runJson.status).toBe("keep"); + }); + + it("rejects keep when an out-of-scope file is dirty", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src/in-scope.ts", + "", + "## Off Limits", + "", + "## Constraints", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 1;\n"); + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 2;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 99;\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src/in-scope.ts"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-6", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail", + asi: { hypothesis: "touch wrong file" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("outside Files in Scope"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("rejects keep when a dirty path is listed under Off Limits", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src", "generated"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 1;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 2;\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-7", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail", + asi: { hypothesis: "touch forbidden path" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Off Limits"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("rejects keep when the metric is worse than the current best kept run", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0003", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 3, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["autoresearch.md"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.state.results = [ + { + runNumber: 1, + commit: "aaaaaaa", + metric: 10, + metrics: {}, + status: "keep", + description: "baseline", + timestamp: 1, + segment: 0, + confidence: null, + }, + { + runNumber: 2, + commit: "bbbbbbb", + metric: 8, + metrics: {}, + status: "keep", + description: "winner", + timestamp: 2, + segment: 0, + confidence: null, + }, + ]; + runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "0003"); + runtime.lastRunNumber = 3; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0003"), + runNumber: 3, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-best", + { + commit: "initial", + metric: 9, + status: "keep", + description: "regression from best", + asi: { hypothesis: "try a weaker variant" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Current best: 8"), + }); + expect(runtime.state.results).toHaveLength(2); + }); + + it("requires failed benchmarks to be logged as crash", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 1, + runNumber: 1, + timedOut: false, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: null, + parsedPrimary: null, + passed: false, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: { + exec: async () => ({ code: 0, stderr: "", stdout: "autoresearch/test-20260323\n" }), + } as unknown as ExtensionAPI, + }); + const result = await tool.execute( + "call-status-crash", + { + commit: "initial", + metric: 0, + status: "discard", + description: "wrong status", + asi: { + hypothesis: "broken attempt", + rollback_reason: "benchmark failed", + next_action_hint: "fix the crash first", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Log it as crash"), + }); + }); + + it("requires failed checks to be logged as checks_failed", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + checks: { durationSeconds: 1, passed: false, timedOut: false }, + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 1, + checksPass: false, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: null, + parsedPrimary: 10, + passed: false, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: { + exec: async () => ({ code: 0, stderr: "", stdout: "autoresearch/test-20260323\n" }), + } as unknown as ExtensionAPI, + }); + const result = await tool.execute( + "call-status-checks", + { + commit: "initial", + metric: 10, + status: "crash", + description: "wrong checks status", + asi: { + hypothesis: "checks regressed", + rollback_reason: "test suite failed", + next_action_hint: "inspect failing checks", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Log it as checks_failed"), + }); + }); + + it("persists autoresearch shutdown when the max iteration cap is reached", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.autoresearchMode = true; + runtime.goal = "reduce runtime"; + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["autoresearch.md"]; + runtime.state.maxExperiments = 1; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "0001"); + runtime.lastRunNumber = 1; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const managedApi = createManagedGitApi({ + activeTools: ["read", "init_experiment", "run_experiment", "log_experiment"], + }); + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: managedApi.api, + }); + const result = await tool.execute( + "call-max", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Baseline", + asi: { hypothesis: "record baseline" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Autoresearch mode is now off"), + }); + expect(runtime.autoresearchMode).toBe(false); + expect(managedApi.appendEntries).toContainEqual({ + customType: "autoresearch-control", + data: { mode: "off", goal: "reduce runtime" }, + }); + expect(managedApi.setActiveToolsCalls).toEqual([["read"]]); + }); + + it("rejects keep when a rename touches an off-limits source path", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const api = { + exec: async (command: string, args: string[]) => { + if (command !== "git") return { code: 1, stderr: "unexpected", stdout: "" }; + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: "R src/generated/index.ts\0src/index.ts\0" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + } as unknown as ExtensionAPI; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: api, + }); + const result = await tool.execute( + "call-8", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail on rename", + asi: { hypothesis: "rename generated file" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Off Limits"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("removes ignored experiment artifacts on discard while preserving autoresearch control files", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + fs.writeFileSync(path.join(dir, ".gitignore"), "tmp-artifact/\n"); + fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.mkdirSync(path.join(dir, "tmp-artifact"), { recursive: true }); + fs.writeFileSync(path.join(dir, "tmp-artifact", "result.txt"), "temporary benchmark output\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 10 }, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 10 }, + parsedPrimary: 10, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-discard", + { + commit: "initial", + metric: 10, + status: "discard", + description: "Discard noisy run", + asi: { + hypothesis: "investigate cache behavior", + rollback_reason: "ignored artifact should be cleaned", + next_action_hint: "try a cleaner setup", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Logged run #1: discard"), + }); + expect(fs.existsSync(path.join(dir, "tmp-artifact"))).toBe(false); + expect(fs.existsSync(path.join(dir, "autoresearch.md"))).toBe(true); + expect(fs.existsSync(path.join(dir, "autoresearch.program.md"))).toBe(true); + }); +});