feat(autoresearch): added auto-resume, path validation, and security guards
- Added auto-resume mechanism with state tracking to automatically resume pending experiment runs and prevent duplicate resumptions. - Added contract path validation to reject unsafe path specifications with absolute paths and parent directory traversal attempts. - Added secondary metrics input to autoresearch setup flow for specifying tradeoff metrics alongside primary objectives. - Enhanced command parsing with shell operator detection to reject piped, redirected, or chained autoresearch.sh commands. - Added prototype pollution guards in object cloning functions to prevent injection via __proto__, constructor, and prototype keys. - Fixed boundary duplication warnings in hashline detection to properly report multiple overlapping hashline references.
This commit is contained in:
@@ -1,7 +1,6 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Changed hashline edit schema from flat `op`/`pos`/`end`/`lines` fields to structured `loc`/`content` format with location-specific objects
|
||||
@@ -14,6 +13,14 @@
|
||||
|
||||
### Added
|
||||
|
||||
- Added prompt for tradeoff metrics during autoresearch setup to collect secondary metrics alongside primary metric
|
||||
- Added validation of contract path specifications to reject absolute paths and parent directory references
|
||||
- Added stricter benchmark command validation in `isAutoresearchShCommand()` to reject chained commands, pipes, and redirects
|
||||
- Added protection against prototype pollution in ASI data and metric cloning by filtering `__proto__`, `constructor`, and `prototype` keys
|
||||
- Added `autoResumeArmed` flag to track when autoresearch should automatically resume pending runs
|
||||
- Added `lastAutoResumePendingRunNumber` to prevent duplicate auto-resume prompts for the same pending run
|
||||
- Added `git clean -X` invocation during failed experiment rollback to remove ignored build artifacts
|
||||
- Added validation to reject `init_experiment` when a previous run is still pending and unlogged
|
||||
- Added autoresearch contract system for validating benchmark commands, metrics, scope paths, off-limits paths, and constraints with fingerprint tracking to detect configuration drift
|
||||
- Added `autoresearch.program.md` support for repo-local playbook overlays that guide session strategy while preserving `autoresearch.md` as source of truth
|
||||
- Added pending run artifact tracking and recovery to resume incomplete experiments from `.autoresearch/runs/` directory with run numbers and benchmark logs
|
||||
@@ -62,6 +69,19 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed `isAutoresearchShCommand()` to use proper command-line argument parsing instead of regex, improving accuracy for complex shell invocations
|
||||
- Changed autoresearch initialization prompt to display collected tradeoff metrics in the setup summary
|
||||
- Changed `command-initialize.md` template to include guidance on preflight requirements, comparability invariants, and marking measurement-critical files as off-limits
|
||||
- Changed `command-initialize.md` to instruct users to write or update `autoresearch.program.md` with durable heuristics and repo-specific strategy
|
||||
- Changed autoresearch resume guidance to emphasize continuing on the current protected branch rather than switching branches
|
||||
- Changed autoresearch prompt to clarify that `autoresearch.md` holds durable conclusions while `autoresearch.ideas.md` is the scratch backlog
|
||||
- Changed autoresearch prompt guidance to require stable measurement harness and fixed benchmark inputs unless intentionally starting a new segment
|
||||
- Changed autoresearch prompt to recommend keeping equal or near-equal results when they materially simplify implementation
|
||||
- Changed `init_experiment` to reset pending run state (checks, duration, ASI, artifact directory) when initializing a new segment
|
||||
- Changed `log_experiment` to set `autoResumeArmed` flag after successfully logging a run to enable auto-resume on next agent turn
|
||||
- Changed `run_experiment` to set `autoResumeArmed` flag and update dashboard after completing a run
|
||||
- Changed auto-resume logic to only prompt when a new pending run exists or when `autoResumeArmed` is explicitly set, preventing duplicate prompts
|
||||
- Changed path normalization in contract validation to use `path.posix.normalize()` for consistent path handling
|
||||
- Changed autoresearch initialization to collect and validate benchmark command, metric definition, scope paths, off-limits list, and constraints before `init_experiment`
|
||||
- Changed `init_experiment` to require exact benchmark command, metric definition, scope, off-limits, and constraints matching collected contract
|
||||
- Changed `log_experiment` to record run number, benchmark command, scope paths, off-limits list, constraints, and segment fingerprint with each result
|
||||
@@ -111,6 +131,9 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed boundary duplication warnings to always display when replacement lines match the next surviving line, even when auto-correction is disabled
|
||||
- Fixed secondary metrics validation to properly reject missing configured metrics and new metrics without force flag
|
||||
- Fixed ASI data cloning to prevent prototype pollution attacks by filtering reserved property names
|
||||
- Fixed autoresearch resume to detect and recover pending run artifacts that were left unlogged from previous sessions
|
||||
- Fixed dashboard overlay to display when running experiment even with zero completed results
|
||||
- Fixed tab character rendering in dashboard command display and tool output summaries
|
||||
|
||||
@@ -10,6 +10,8 @@ Collected setup:
|
||||
- primary metric: `{{metric_name}}`
|
||||
- metric unit: `{{metric_unit}}`
|
||||
- direction: `{{direction}}`
|
||||
- tradeoff metrics:
|
||||
{{{secondary_metrics_block}}}
|
||||
- files in scope:
|
||||
{{{scope_paths_block}}}
|
||||
- off limits:
|
||||
@@ -21,8 +23,10 @@ Explain briefly what autoresearch will do in this repository, then initialize th
|
||||
|
||||
Your first actions:
|
||||
- write `autoresearch.md`
|
||||
- record the collected benchmark command, primary metric, metric unit, direction, scope, off-limits list, and constraints in `autoresearch.md`
|
||||
- optionally write `autoresearch.program.md` when a repo-local playbook would help future resume quality
|
||||
- record the collected benchmark command, primary metric, metric unit, direction, tradeoff metrics, scope, off-limits list, and constraints in `autoresearch.md`
|
||||
- add a short preflight section in `autoresearch.md` covering prerequisites, one-time setup, and the comparability invariant that must stay fixed across runs
|
||||
- explicitly mark the ground-truth evaluator, fixed datasets, and other measurement-critical files as off-limits or hard constraints when they define the benchmark contract
|
||||
- write or update `autoresearch.program.md` when you learn durable heuristics, failure patterns, or repo-specific strategy that future resume turns should inherit
|
||||
- define the benchmark entrypoint in `autoresearch.sh`
|
||||
- optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate
|
||||
- run `init_experiment` with the exact collected benchmark command, metric definition, scope paths, off-limits list, and constraints
|
||||
|
||||
@@ -13,5 +13,5 @@ Additional context from the user:
|
||||
Use the notes as the source of truth for the current direction, scope, and constraints.
|
||||
- inspect recent git history for context
|
||||
- inspect `autoresearch.jsonl` if it exists
|
||||
- continue the most promising unfinished branch
|
||||
- continue the most promising unfinished direction on the current protected branch
|
||||
- keep iterating until interrupted or until the configured iteration cap is reached
|
||||
|
||||
@@ -63,6 +63,16 @@ export function validateAutoresearchContract(contract: AutoresearchContract): st
|
||||
if (contract.scopePaths.length === 0) {
|
||||
errors.push("Files in Scope must contain at least one path in autoresearch.md.");
|
||||
}
|
||||
for (const scopePath of contract.scopePaths) {
|
||||
if (isUnsafeContractPathSpec(scopePath)) {
|
||||
errors.push(`Files in Scope contains an invalid path: ${scopePath}`);
|
||||
}
|
||||
}
|
||||
for (const offLimitsPath of contract.offLimits) {
|
||||
if (isUnsafeContractPathSpec(offLimitsPath)) {
|
||||
errors.push(`Off Limits contains an invalid path: ${offLimitsPath}`);
|
||||
}
|
||||
}
|
||||
return errors;
|
||||
}
|
||||
|
||||
@@ -151,7 +161,7 @@ export function normalizeAutoresearchList(values: readonly string[]): string[] {
|
||||
}
|
||||
|
||||
export function normalizeContractPathSpec(value: string): string {
|
||||
const normalized = value.trim().replaceAll("\\", "/");
|
||||
const normalized = path.posix.normalize(value.trim().replaceAll("\\", "/"));
|
||||
if (normalized === "." || normalized === "./") return ".";
|
||||
return normalized.replace(/^\.\/+/, "").replace(/\/+$/, "");
|
||||
}
|
||||
@@ -316,3 +326,7 @@ function parseSecondaryMetrics(value: string | undefined): string[] {
|
||||
.filter(Boolean),
|
||||
);
|
||||
}
|
||||
|
||||
function isUnsafeContractPathSpec(value: string): boolean {
|
||||
return path.posix.isAbsolute(value) || value === ".." || value.startsWith("../");
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { isEnoent } from "@oh-my-pi/pi-utils";
|
||||
import { parseCommandArgs } from "../utils/command-args";
|
||||
import type {
|
||||
ASIData,
|
||||
ASIValue,
|
||||
@@ -185,8 +186,36 @@ export function isAutoresearchShCommand(command: string): boolean {
|
||||
previous = normalized;
|
||||
normalized = normalized.replace(/^(?:env|time|nice|nohup)(?:\s+-\S+(?:\s+\d+)?)?\s+/, "");
|
||||
}
|
||||
if (/[;&|<>]/.test(normalized)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return /^(?:(?:bash|sh)\s+(?:-\w+\s+)*)?(?:\.\/|\/[\w/.-]*\/)?autoresearch\.sh(?:\s|$)/.test(normalized);
|
||||
const tokens = parseCommandArgs(normalized);
|
||||
if (tokens.length === 0) return false;
|
||||
|
||||
let index = 0;
|
||||
if (tokens[index] === "bash" || tokens[index] === "sh") {
|
||||
index += 1;
|
||||
while (index < tokens.length && tokens[index]?.startsWith("-")) {
|
||||
if (tokens[index]?.includes("c")) {
|
||||
return false;
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
}
|
||||
|
||||
const scriptToken = tokens[index];
|
||||
if (!scriptToken || !/^(?:\.\/|\/[\w/.-]*\/)?autoresearch\.sh$/.test(scriptToken)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (const token of tokens.slice(index + 1)) {
|
||||
if (token === "&&" || token === "||" || token === ";" || token === "|" || token === ">" || token === "<") {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
export function isBetter(current: number, best: number, direction: MetricDirection): boolean {
|
||||
@@ -380,6 +409,7 @@ function cloneNumericMetricMap(value: unknown): NumericMetricMap | null {
|
||||
const metrics = value as { [key: string]: unknown };
|
||||
const clone: NumericMetricMap = {};
|
||||
for (const [key, entryValue] of Object.entries(metrics)) {
|
||||
if (DENIED_KEY_NAMES.has(key)) continue;
|
||||
if (typeof entryValue === "number" && Number.isFinite(entryValue)) {
|
||||
clone[key] = entryValue;
|
||||
}
|
||||
@@ -392,6 +422,7 @@ function cloneAsiData(value: unknown): ASIData | null {
|
||||
const candidate = value as { [key: string]: unknown };
|
||||
const clone: ASIData = {};
|
||||
for (const [key, entryValue] of Object.entries(candidate)) {
|
||||
if (DENIED_KEY_NAMES.has(key)) continue;
|
||||
const sanitized = clonePendingAsiValue(entryValue);
|
||||
if (sanitized !== undefined) {
|
||||
clone[key] = sanitized;
|
||||
@@ -415,6 +446,7 @@ function clonePendingAsiValue(value: unknown): ASIValue | undefined {
|
||||
const candidate = value as { [key: string]: unknown };
|
||||
const clone: { [key: string]: ASIValue } = {};
|
||||
for (const [key, entryValue] of Object.entries(candidate)) {
|
||||
if (DENIED_KEY_NAMES.has(key)) continue;
|
||||
const sanitized = clonePendingAsiValue(entryValue);
|
||||
if (sanitized !== undefined) {
|
||||
clone[key] = sanitized;
|
||||
|
||||
@@ -43,6 +43,7 @@ interface AutoresearchSetupInput {
|
||||
metricName: string;
|
||||
metricUnit: string;
|
||||
direction: "lower" | "higher";
|
||||
secondaryMetrics: string[];
|
||||
scopePaths: string[];
|
||||
offLimits: string[];
|
||||
constraints: string[];
|
||||
@@ -65,6 +66,8 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
runtime.state.maxExperiments = readMaxExperiments(ctx.cwd);
|
||||
runtime.goal = control.goal;
|
||||
runtime.autoresearchMode = control.autoresearchMode;
|
||||
runtime.autoResumeArmed = false;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
runtime.lastRunSummary = await readPendingRunSummary(workDir, loggedRunNumbers);
|
||||
runtime.lastRunChecks = summaryToChecks(runtime.lastRunSummary);
|
||||
runtime.lastRunDuration = runtime.lastRunSummary?.durationSeconds ?? null;
|
||||
@@ -94,7 +97,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
): void => {
|
||||
const runtime = getRuntime(ctx);
|
||||
runtime.autoresearchMode = enabled;
|
||||
runtime.autoResumeArmed = false;
|
||||
runtime.goal = goal;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
api.appendEntry("autoresearch-control", goal ? { mode, goal } : { mode });
|
||||
};
|
||||
|
||||
@@ -251,6 +256,7 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
runtime.state.metricName = setup.metricName;
|
||||
runtime.state.metricUnit = setup.metricUnit;
|
||||
runtime.state.bestDirection = setup.direction;
|
||||
runtime.state.secondaryMetrics = setup.secondaryMetrics.map(name => ({ name, unit: "" }));
|
||||
runtime.state.benchmarkCommand = setup.benchmarkCommand;
|
||||
runtime.state.scopePaths = [...setup.scopePaths];
|
||||
runtime.state.offLimits = [...setup.offLimits];
|
||||
@@ -267,6 +273,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
metric_name: setup.metricName,
|
||||
metric_unit: setup.metricUnit,
|
||||
direction: setup.direction,
|
||||
has_secondary_metrics: setup.secondaryMetrics.length > 0,
|
||||
secondary_metrics: setup.secondaryMetrics,
|
||||
secondary_metrics_block: formatBulletBlock(
|
||||
setup.secondaryMetrics,
|
||||
value => ` - \`${value}\``,
|
||||
" - `(none)`",
|
||||
),
|
||||
scope_paths: setup.scopePaths,
|
||||
scope_paths_block: formatBulletBlock(setup.scopePaths, value => ` - \`${value}\``),
|
||||
has_off_limits: setup.offLimits.length > 0,
|
||||
@@ -315,7 +328,10 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
dashboard.updateWidget(ctx, runtime);
|
||||
dashboard.requestRender();
|
||||
if (!runtime.autoresearchMode) return;
|
||||
if (ctx.hasPendingMessages()) return;
|
||||
if (ctx.hasPendingMessages()) {
|
||||
runtime.autoResumeArmed = false;
|
||||
return;
|
||||
}
|
||||
const workDir = resolveWorkDir(ctx.cwd);
|
||||
const pendingRun =
|
||||
runtime.lastRunSummary ??
|
||||
@@ -324,6 +340,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => {
|
||||
runtime.lastRunChecks = summaryToChecks(pendingRun);
|
||||
runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration;
|
||||
runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi;
|
||||
const shouldResumePendingRun =
|
||||
pendingRun !== null && runtime.lastAutoResumePendingRunNumber !== pendingRun.runNumber;
|
||||
if (!shouldResumePendingRun && !runtime.autoResumeArmed) {
|
||||
return;
|
||||
}
|
||||
runtime.autoResumeArmed = false;
|
||||
runtime.lastAutoResumePendingRunNumber = pendingRun?.runNumber ?? null;
|
||||
const autoresearchMdPath = path.join(workDir, "autoresearch.md");
|
||||
const ideasPath = path.join(workDir, "autoresearch.ideas.md");
|
||||
api.sendMessage(
|
||||
@@ -459,6 +482,9 @@ async function promptForAutoresearchSetup(
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const secondaryMetricsInput = await ctx.ui.input("Tradeoff Metrics", "");
|
||||
if (secondaryMetricsInput === undefined) return undefined;
|
||||
|
||||
const scopePathsInput = await ctx.ui.input("Files in Scope", "packages/coding-agent/src/autoresearch");
|
||||
if (scopePathsInput === undefined) return undefined;
|
||||
const scopePaths = splitSetupList(scopePathsInput);
|
||||
@@ -478,6 +504,7 @@ async function promptForAutoresearchSetup(
|
||||
metricName,
|
||||
metricUnit,
|
||||
direction: normalizedDirection,
|
||||
secondaryMetrics: splitSetupList(secondaryMetricsInput),
|
||||
scopePaths,
|
||||
offLimits: splitSetupList(offLimitsInput),
|
||||
constraints: splitSetupList(constraintsInput),
|
||||
|
||||
@@ -71,13 +71,17 @@ An unlogged run artifact exists at `{{pending_run_directory}}`.
|
||||
- Read the relevant source files.
|
||||
- Identify the true bottleneck or quality constraint.
|
||||
- Check existing scripts, benchmark harnesses, and config files.
|
||||
- Verify prerequisites, one-time setup, and benchmark inputs before the first run of a segment.
|
||||
2. Keep your notes in `autoresearch.md`.
|
||||
- Record the goal, the benchmark command, the primary metric, important secondary metrics, the files in scope, hard constraints, and the running ideas backlog.
|
||||
- Record the goal, the benchmark command, the primary metric, important secondary metrics, the files in scope, hard constraints, preflight requirements, and the benchmark comparability invariant.
|
||||
- Update the notes whenever the strategy changes.
|
||||
- Keep durable conclusions in `autoresearch.md`.
|
||||
- Use `autoresearch.ideas.md` for deferred experiment ideas that are promising but not active yet.
|
||||
3. Use `autoresearch.sh` as the canonical benchmark entrypoint.
|
||||
- If it does not exist yet, create it.
|
||||
- Make it print structured metric lines in the form `METRIC name=value`.
|
||||
- Use the same workload every run unless you intentionally re-initialize with a new segment.
|
||||
- Keep the measurement harness, evaluator, and fixed benchmark inputs stable unless you intentionally start a new segment and document the change.
|
||||
4. Initialize the loop with `init_experiment` before the first logged run of a segment.
|
||||
5. Run a baseline first.
|
||||
- Establish the baseline metric before attempting optimizations.
|
||||
@@ -98,7 +102,8 @@ An unlogged run artifact exists at `{{pending_run_directory}}`.
|
||||
- Use ASI to capture what you learned, not just what you changed.
|
||||
9. Prefer simpler wins.
|
||||
- Remove dead ends.
|
||||
- Do not keep complexity that does not move the metric.
|
||||
- Keep equal or near-equal results when they materially simplify the implementation.
|
||||
- Do not keep ugly complexity for tiny gains unless the payoff is clearly worth it.
|
||||
- Do not thrash between unrelated ideas without writing down the conclusion.
|
||||
10. When confidence is low, confirm.
|
||||
- The dashboard confidence score compares the best observed improvement against the observed noise floor.
|
||||
@@ -116,6 +121,8 @@ Your benchmark script SHOULD:
|
||||
- print secondary metrics as additional `METRIC name=value` lines
|
||||
- avoid extra randomness when possible
|
||||
- use repeated samples and median-style summaries for fast benchmarks
|
||||
- preserve the comparability invariant for the current segment
|
||||
- keep the ground-truth evaluator and fixed benchmark inputs unchanged unless the segment is explicitly re-initialized
|
||||
|
||||
### Notes file template
|
||||
|
||||
@@ -182,7 +189,7 @@ Resume from the existing notes:
|
||||
- read `autoresearch.md`
|
||||
- inspect recent git history
|
||||
- inspect `autoresearch.jsonl`
|
||||
- continue from the most promising unfinished branch
|
||||
- continue from the most promising unfinished direction on the current protected branch
|
||||
|
||||
{{else}}
|
||||
### Initial setup
|
||||
@@ -215,6 +222,6 @@ Treat failing checks as a failed experiment:
|
||||
|
||||
`autoresearch.ideas.md` exists at `{{ideas_path}}`.
|
||||
|
||||
Use it to keep promising but deferred experiments. Prune stale ideas when they are disproven or superseded.
|
||||
Use it to keep promising but deferred experiments. `autoresearch.md` should hold durable conclusions; `autoresearch.ideas.md` is the scratch backlog. Prune stale ideas when they are disproven or superseded.
|
||||
|
||||
{{/if}}
|
||||
|
||||
@@ -10,7 +10,7 @@ Continue the autoresearch loop now.
|
||||
{{/if}}
|
||||
- Continue from the most promising unfinished direction.
|
||||
{{#if has_ideas}}
|
||||
- Review `autoresearch.ideas.md` for promising next steps and prune stale items.
|
||||
- Review `autoresearch.ideas.md` for deferred next steps and prune stale items.
|
||||
{{/if}}
|
||||
- Keep iterating until interrupted or until the configured iteration cap is reached.
|
||||
- Preserve correctness and do not game the benchmark.
|
||||
|
||||
@@ -41,7 +41,9 @@ export function createExperimentState(): ExperimentState {
|
||||
export function createSessionRuntime(): AutoresearchRuntime {
|
||||
return {
|
||||
autoresearchMode: false,
|
||||
autoResumeArmed: false,
|
||||
dashboardExpanded: false,
|
||||
lastAutoResumePendingRunNumber: null,
|
||||
lastRunChecks: null,
|
||||
lastRunDuration: null,
|
||||
lastRunAsi: null,
|
||||
@@ -341,6 +343,7 @@ function cloneNumericMetrics(value: unknown): NumericMetricMap {
|
||||
const metrics = value as { [key: string]: unknown };
|
||||
const clone: NumericMetricMap = {};
|
||||
for (const [key, entryValue] of Object.entries(metrics)) {
|
||||
if (key === "__proto__" || key === "constructor" || key === "prototype") continue;
|
||||
if (typeof entryValue === "number" && Number.isFinite(entryValue)) {
|
||||
clone[key] = entryValue;
|
||||
}
|
||||
@@ -363,7 +366,12 @@ function hydrateMetricDefs(metricNames: string[] | undefined): MetricDef[] {
|
||||
|
||||
function cloneAsi(value: unknown): ExperimentResult["asi"] {
|
||||
if (typeof value !== "object" || value === null) return undefined;
|
||||
return structuredClone(value) as ExperimentResult["asi"];
|
||||
const clone: { [key: string]: unknown } = {};
|
||||
for (const [key, entryValue] of Object.entries(value)) {
|
||||
if (key === "__proto__" || key === "constructor" || key === "prototype") continue;
|
||||
clone[key] = structuredClone(entryValue);
|
||||
}
|
||||
return clone as ExperimentResult["asi"];
|
||||
}
|
||||
|
||||
function parseControlEntry(value: unknown): AutoresearchControlEntryData | null {
|
||||
|
||||
@@ -17,6 +17,7 @@ import {
|
||||
inferMetricUnitFromName,
|
||||
isAutoresearchShCommand,
|
||||
readMaxExperiments,
|
||||
readPendingRunSummary,
|
||||
resolveWorkDir,
|
||||
validateWorkDir,
|
||||
} from "../helpers";
|
||||
@@ -85,6 +86,19 @@ export function createInitExperimentTool(
|
||||
const state = runtime.state;
|
||||
const isReinitializing = state.results.length > 0;
|
||||
const workDir = resolveWorkDir(ctx.cwd);
|
||||
const pendingRun = await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results));
|
||||
if (pendingRun) {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text:
|
||||
`Error: run #${pendingRun.runNumber} has not been logged yet. ` +
|
||||
"Call log_experiment before re-initializing the current segment.",
|
||||
},
|
||||
],
|
||||
};
|
||||
}
|
||||
const contractResult = readAutoresearchContract(workDir);
|
||||
const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir);
|
||||
const errors = [...contractResult.errors, ...scriptSnapshot.errors];
|
||||
@@ -241,6 +255,14 @@ export function createInitExperimentTool(
|
||||
}
|
||||
|
||||
runtime.autoresearchMode = true;
|
||||
runtime.autoResumeArmed = true;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
runtime.lastRunChecks = null;
|
||||
runtime.lastRunDuration = null;
|
||||
runtime.lastRunAsi = null;
|
||||
runtime.lastRunArtifactDir = null;
|
||||
runtime.lastRunNumber = null;
|
||||
runtime.lastRunSummary = null;
|
||||
options.dashboard.updateWidget(ctx, runtime);
|
||||
options.dashboard.requestRender();
|
||||
|
||||
@@ -276,3 +298,13 @@ export function createInitExperimentTool(
|
||||
function renderInitCall(name: string, theme: Theme): string {
|
||||
return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", truncateToWidth(replaceTabs(name), 100))}`;
|
||||
}
|
||||
|
||||
function collectLoggedRunNumbers(results: ExperimentState["results"]): Set<number> {
|
||||
const runNumbers = new Set<number>();
|
||||
for (const result of results) {
|
||||
if (result.runNumber !== null) {
|
||||
runNumbers.add(result.runNumber);
|
||||
}
|
||||
}
|
||||
return runNumbers;
|
||||
}
|
||||
|
||||
@@ -304,6 +304,8 @@ export function createLogExperimentTool(
|
||||
runtime.lastRunArtifactDir = null;
|
||||
runtime.lastRunNumber = null;
|
||||
runtime.lastRunSummary = null;
|
||||
runtime.autoResumeArmed = true;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
|
||||
const currentSegmentRuns = currentResults(state.results, state.currentSegment).length;
|
||||
const text = buildLogText(state, experiment, currentSegmentRuns, wallClockSeconds, gitNote);
|
||||
@@ -364,10 +366,12 @@ function buildSecondaryMetrics(
|
||||
): NumericMetricMap {
|
||||
const merged: NumericMetricMap = {};
|
||||
for (const [name, value] of Object.entries(parsedMetrics ?? {})) {
|
||||
if (name === "__proto__" || name === "constructor" || name === "prototype") continue;
|
||||
if (name === primaryMetricName) continue;
|
||||
merged[name] = value;
|
||||
}
|
||||
for (const [name, value] of Object.entries(cloneMetrics(overrides))) {
|
||||
if (name === "__proto__" || name === "constructor" || name === "prototype") continue;
|
||||
merged[name] = value;
|
||||
}
|
||||
return merged;
|
||||
@@ -377,6 +381,7 @@ function sanitizeAsi(value: { [key: string]: unknown } | undefined): ASIData | u
|
||||
if (!value) return undefined;
|
||||
const result: ASIData = {};
|
||||
for (const [key, entryValue] of Object.entries(value)) {
|
||||
if (key === "__proto__" || key === "constructor" || key === "prototype") continue;
|
||||
const sanitized = sanitizeAsiValue(entryValue);
|
||||
if (sanitized !== undefined) {
|
||||
result[key] = sanitized;
|
||||
@@ -398,6 +403,7 @@ function sanitizeAsiValue(value: unknown): ASIData[string] | undefined {
|
||||
const objectValue = value as { [key: string]: unknown };
|
||||
const result: ASIData = {};
|
||||
for (const [key, entryValue] of Object.entries(objectValue)) {
|
||||
if (key === "__proto__" || key === "constructor" || key === "prototype") continue;
|
||||
const sanitized = sanitizeAsiValue(entryValue);
|
||||
if (sanitized !== undefined) {
|
||||
result[key] = sanitized;
|
||||
@@ -567,6 +573,10 @@ async function revertFailedExperiment(
|
||||
{ cwd: workDir, timeout: 10_000 },
|
||||
);
|
||||
const cleanResult = await options.pi.exec("git", ["clean", "-fd", "--", "."], { cwd: workDir, timeout: 10_000 });
|
||||
const cleanIgnoredResult = await options.pi.exec("git", ["clean", "-fdX", "--", "."], {
|
||||
cwd: workDir,
|
||||
timeout: 10_000,
|
||||
});
|
||||
restoreAutoresearchFiles(preservedFiles);
|
||||
if (restoreResult.code !== 0) {
|
||||
return {
|
||||
@@ -578,6 +588,11 @@ async function revertFailedExperiment(
|
||||
error: `git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`,
|
||||
};
|
||||
}
|
||||
if (cleanIgnoredResult.code !== 0) {
|
||||
return {
|
||||
error: `git clean -X failed: ${mergeStdoutStderr(cleanIgnoredResult).trim() || `exit ${cleanIgnoredResult.code}`}`,
|
||||
};
|
||||
}
|
||||
const dirtyCheckResult = await options.pi.exec(
|
||||
"git",
|
||||
["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."],
|
||||
|
||||
@@ -303,6 +303,10 @@ export function createRunExperimentTool(
|
||||
runDirectory,
|
||||
runNumber,
|
||||
};
|
||||
runtime.autoResumeArmed = true;
|
||||
runtime.lastAutoResumePendingRunNumber = null;
|
||||
options.dashboard.updateWidget(ctx, runtime);
|
||||
options.dashboard.requestRender();
|
||||
|
||||
await Bun.write(
|
||||
runJsonPath,
|
||||
|
||||
@@ -135,7 +135,9 @@ export interface RunningExperiment {
|
||||
|
||||
export interface AutoresearchRuntime {
|
||||
autoresearchMode: boolean;
|
||||
autoResumeArmed: boolean;
|
||||
dashboardExpanded: boolean;
|
||||
lastAutoResumePendingRunNumber: number | null;
|
||||
lastRunChecks: ChecksResult | null;
|
||||
lastRunDuration: number | null;
|
||||
lastRunAsi: ASIData | null;
|
||||
|
||||
@@ -567,7 +567,7 @@ export function applyHashlineEdits(
|
||||
const tag = formatLineTag(endLine + 1, nextSurvivingLine);
|
||||
warnings.push(
|
||||
`Possible boundary duplication: your last replacement line \`${trimmedLast}\` is identical to the next surviving line ${tag}. ` +
|
||||
`If you meant to replace the entire block, set \`end\` to ${tag} instead.`,
|
||||
`If you meant to replace the entire block, set \`end\` to ${tag} instead.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,4 +114,4 @@ When adding a sibling declaration, prefer `prepend` on the next declaration.
|
||||
- For a block, either replace only the body or replace the whole block. Do not split block boundaries.
|
||||
- `content` must be literal file content with matching indentation. If the file uses tabs, use real tabs.
|
||||
- Do not use this tool to reformat or clean up unrelated code.
|
||||
</critical>
|
||||
</critical>
|
||||
@@ -6,8 +6,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { Agent, AgentBusyError, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import { type AssistantMessage, getBundledModel, type ToolCall } from "@oh-my-pi/pi-ai";
|
||||
import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import { type AssistantMessage, getBundledModel, type Message, type ToolCall } from "@oh-my-pi/pi-ai";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
@@ -15,6 +15,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr";
|
||||
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import { Type } from "@sinclair/typebox";
|
||||
@@ -112,6 +113,16 @@ describe("AgentSession concurrent prompt guard", () => {
|
||||
return session;
|
||||
}
|
||||
|
||||
async function waitFor(predicate: () => boolean, timeoutMs = 500): Promise<void> {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (Date.now() < deadline) {
|
||||
if (predicate()) return;
|
||||
await Bun.sleep(10);
|
||||
}
|
||||
|
||||
throw new Error("Timed out waiting for condition");
|
||||
}
|
||||
|
||||
it("should throw when prompt() called while streaming", async () => {
|
||||
await createSession();
|
||||
|
||||
@@ -167,7 +178,7 @@ describe("AgentSession concurrent prompt guard", () => {
|
||||
it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => {
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
|
||||
let firstStream: MockAssistantStream | undefined;
|
||||
const callMessages: AgentMessage[][] = [];
|
||||
const callMessages: Message[][] = [];
|
||||
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
@@ -176,6 +187,7 @@ describe("AgentSession concurrent prompt guard", () => {
|
||||
systemPrompt: "Test",
|
||||
tools: [],
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (_model, context) => {
|
||||
callMessages.push([...context.messages]);
|
||||
const stream = new MockAssistantStream();
|
||||
@@ -206,7 +218,7 @@ describe("AgentSession concurrent prompt guard", () => {
|
||||
});
|
||||
|
||||
const firstPrompt = session.prompt("First message");
|
||||
await Bun.sleep(10);
|
||||
await waitFor(() => session.isStreaming && firstStream !== undefined && callMessages.length === 1);
|
||||
|
||||
await session.sendCustomMessage(
|
||||
{
|
||||
@@ -227,10 +239,15 @@ describe("AgentSession concurrent prompt guard", () => {
|
||||
|
||||
expect(callMessages).toHaveLength(2);
|
||||
expect(
|
||||
callMessages[1]?.some(
|
||||
message =>
|
||||
message.role === "custom" && "customType" in message && message.customType === "autoresearch-resume",
|
||||
),
|
||||
callMessages[1]?.some(message => {
|
||||
if (typeof message.content === "string") {
|
||||
return message.content.includes("Hidden stop reaction");
|
||||
}
|
||||
|
||||
return message.content.some(
|
||||
content => content.type === "text" && content.text.includes("Hidden stop reaction"),
|
||||
);
|
||||
}),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
|
||||
@@ -297,6 +297,12 @@ describe("autoresearch command guard", () => {
|
||||
expect(isAutoresearchShCommand("echo hi; autoresearch.sh")).toBe(false);
|
||||
expect(isAutoresearchShCommand("bash -lc 'autoresearch.sh'")).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects chained or redirected benchmark commands even when autoresearch.sh comes first", () => {
|
||||
expect(isAutoresearchShCommand("bash autoresearch.sh && touch /tmp/marker")).toBe(false);
|
||||
expect(isAutoresearchShCommand("./autoresearch.sh | tee run.log")).toBe(false);
|
||||
expect(isAutoresearchShCommand("./autoresearch.sh > run.log")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
interface AutoresearchCommandHarness {
|
||||
@@ -394,6 +400,9 @@ function createAutoresearchCommandHarness(
|
||||
}
|
||||
|
||||
interface AutoresearchLifecycleHarness {
|
||||
beforeAgentStartHandler:
|
||||
| ((event: { systemPrompt: string }, ctx: ExtensionContext) => Promise<unknown> | unknown)
|
||||
| undefined;
|
||||
sessionStartHandler: ((event: SessionStartEvent, ctx: ExtensionContext) => Promise<void> | void) | undefined;
|
||||
sessionSwitchHandler: ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise<void> | void) | undefined;
|
||||
agentEndHandler: ((event: unknown, ctx: ExtensionContext) => Promise<void> | void) | undefined;
|
||||
@@ -475,6 +484,9 @@ function createAutoresearchLifecycleHarness(options: {
|
||||
} as unknown as ExtensionContext;
|
||||
|
||||
return {
|
||||
beforeAgentStartHandler: handlers.get("before_agent_start") as
|
||||
| ((event: { systemPrompt: string }, ctx: ExtensionContext) => Promise<unknown> | unknown)
|
||||
| undefined,
|
||||
sessionStartHandler: handlers.get("session_start") as
|
||||
| ((event: SessionStartEvent, ctx: ExtensionContext) => Promise<void> | void)
|
||||
| undefined,
|
||||
@@ -515,6 +527,7 @@ describe("autoresearch command startup", () => {
|
||||
"runtime_ms",
|
||||
"ms",
|
||||
"lower",
|
||||
"memory_mb, rss_mb",
|
||||
"packages/coding-agent/src/autoresearch, packages/coding-agent/test",
|
||||
"packages/coding-agent/src/generated",
|
||||
"preserve output format",
|
||||
@@ -547,6 +560,7 @@ describe("autoresearch command startup", () => {
|
||||
{ title: "Primary Metric Name", placeholder: "runtime_ms" },
|
||||
{ title: "Metric Unit", placeholder: "ms" },
|
||||
{ title: "Metric Direction", placeholder: "lower" },
|
||||
{ title: "Tradeoff Metrics", placeholder: "" },
|
||||
{ title: "Files in Scope", placeholder: "packages/coding-agent/src/autoresearch" },
|
||||
{ title: "Off Limits", placeholder: "" },
|
||||
{ title: "Constraints", placeholder: "" },
|
||||
@@ -558,6 +572,8 @@ describe("autoresearch command startup", () => {
|
||||
expect(harness.sentMessages[0]).toContain("primary metric: `runtime_ms`");
|
||||
expect(harness.sentMessages[0]).toContain("metric unit: `ms`");
|
||||
expect(harness.sentMessages[0]).toContain("direction: `lower`");
|
||||
expect(harness.sentMessages[0]).toContain("`memory_mb`");
|
||||
expect(harness.sentMessages[0]).toContain("`rss_mb`");
|
||||
expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/autoresearch`");
|
||||
expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/generated`");
|
||||
expect(harness.sentMessages[0]).toContain("preserve output format");
|
||||
@@ -587,21 +603,17 @@ describe("autoresearch command startup", () => {
|
||||
await harness.command.handler("", harness.ctx);
|
||||
|
||||
expect(harness.inputCalls).toEqual([]);
|
||||
expect(harness.sentMessages).toEqual([
|
||||
[
|
||||
"Resume autoresearch from the attached notes.",
|
||||
"",
|
||||
`@${autoresearchMdPath}`,
|
||||
"",
|
||||
"Using dedicated git branch `autoresearch/existing-20260322`.",
|
||||
"",
|
||||
"Use the notes as the source of truth for the current direction, scope, and constraints.",
|
||||
"- inspect recent git history for context",
|
||||
"- inspect `autoresearch.jsonl` if it exists",
|
||||
"- continue the most promising unfinished branch",
|
||||
"- keep iterating until interrupted or until the configured iteration cap is reached",
|
||||
].join("\n"),
|
||||
]);
|
||||
expect(harness.sentMessages).toHaveLength(1);
|
||||
expect(harness.sentMessages[0]).toContain("Resume autoresearch from the attached notes.");
|
||||
expect(harness.sentMessages[0]).toContain(`@${autoresearchMdPath}`);
|
||||
expect(harness.sentMessages[0]).toContain("Using dedicated git branch `autoresearch/existing-20260322`.");
|
||||
expect(harness.sentMessages[0]).toContain(
|
||||
"Use the notes as the source of truth for the current direction, scope, and constraints.",
|
||||
);
|
||||
expect(harness.sentMessages[0]).toContain("- inspect `autoresearch.jsonl` if it exists");
|
||||
expect(harness.sentMessages[0]).toContain(
|
||||
"- continue the most promising unfinished direction on the current protected branch",
|
||||
);
|
||||
});
|
||||
|
||||
it("includes explicit resume context when the user resumes with additional instructions", async () => {
|
||||
@@ -642,6 +654,7 @@ describe("autoresearch command startup", () => {
|
||||
"runtime_ms",
|
||||
"ms",
|
||||
"lower",
|
||||
"",
|
||||
"packages/coding-agent/src/autoresearch",
|
||||
"",
|
||||
"",
|
||||
@@ -741,12 +754,16 @@ describe("autoresearch command startup", () => {
|
||||
"runtime_ms",
|
||||
"ms",
|
||||
"lower",
|
||||
"",
|
||||
"packages/coding-agent/src/autoresearch",
|
||||
"",
|
||||
"",
|
||||
],
|
||||
async (command, args) => {
|
||||
if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" };
|
||||
if (args[0] === "rev-parse" && args[1] === "--show-prefix") {
|
||||
return { code: 0, stderr: "", stdout: "" };
|
||||
}
|
||||
if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` };
|
||||
if (args[0] === "branch" && args[1] === "--show-current") {
|
||||
return { code: 0, stderr: "", stdout: "main\n" };
|
||||
@@ -782,12 +799,16 @@ describe("autoresearch command startup", () => {
|
||||
"runtime_ms",
|
||||
"ms",
|
||||
"lower",
|
||||
"",
|
||||
"packages/coding-agent/src/autoresearch",
|
||||
"",
|
||||
"",
|
||||
],
|
||||
async (command, args) => {
|
||||
if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" };
|
||||
if (args[0] === "rev-parse" && args[1] === "--show-prefix") {
|
||||
return { code: 0, stderr: "", stdout: "" };
|
||||
}
|
||||
if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` };
|
||||
if (args[0] === "branch" && args[1] === "--show-current") {
|
||||
return { code: 0, stderr: "", stdout: "main\n" };
|
||||
@@ -814,6 +835,7 @@ describe("autoresearch command startup", () => {
|
||||
"runtime_ms",
|
||||
"ms",
|
||||
"lower",
|
||||
"",
|
||||
"packages/coding-agent/src/autoresearch",
|
||||
"",
|
||||
"",
|
||||
@@ -1057,6 +1079,99 @@ describe("autoresearch auto-resume", () => {
|
||||
triggerTurn: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("does not enqueue another hidden turn after a passive autoresearch turn with no pending run", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
fs.writeFileSync(
|
||||
path.join(dir, "autoresearch.jsonl"),
|
||||
`${JSON.stringify({ type: "config", metricName: "runtime_ms", scopePaths: ["src"] })}\n`,
|
||||
);
|
||||
|
||||
const harness = createAutoresearchLifecycleHarness({
|
||||
activeTools: ["init_experiment", "run_experiment", "log_experiment"],
|
||||
controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }],
|
||||
cwd: dir,
|
||||
});
|
||||
|
||||
await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx);
|
||||
await harness.agentEndHandler?.({}, harness.ctx);
|
||||
|
||||
expect(harness.sentMessages).toEqual([]);
|
||||
});
|
||||
|
||||
it("renders the high-signal prompt sections for playbooks, backlog, recent runs, and pending runs", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
fs.writeFileSync(path.join(dir, "autoresearch.md"), "# Autoresearch\n");
|
||||
fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Local Playbook\n");
|
||||
fs.writeFileSync(path.join(dir, "autoresearch.ideas.md"), "- try batching\n");
|
||||
fs.writeFileSync(path.join(dir, "autoresearch.checks.sh"), "#!/usr/bin/env bash\n");
|
||||
fs.writeFileSync(
|
||||
path.join(dir, "autoresearch.jsonl"),
|
||||
[
|
||||
JSON.stringify({
|
||||
type: "config",
|
||||
metricName: "runtime_ms",
|
||||
metricUnit: "ms",
|
||||
scopePaths: ["src"],
|
||||
}),
|
||||
JSON.stringify({
|
||||
run: 1,
|
||||
commit: "aaaaaaa",
|
||||
metric: 10,
|
||||
status: "keep",
|
||||
description: "baseline",
|
||||
timestamp: 1,
|
||||
asi: { hypothesis: "baseline" },
|
||||
}),
|
||||
JSON.stringify({
|
||||
run: 2,
|
||||
commit: "bbbbbbb",
|
||||
metric: 9,
|
||||
status: "discard",
|
||||
description: "too noisy",
|
||||
timestamp: 2,
|
||||
asi: {
|
||||
hypothesis: "raise cache size",
|
||||
rollback_reason: "noise",
|
||||
next_action_hint: "re-test with more samples",
|
||||
},
|
||||
}),
|
||||
].join("\n"),
|
||||
);
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0003", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedPrimary: 8,
|
||||
runNumber: 3,
|
||||
}),
|
||||
);
|
||||
|
||||
const harness = createAutoresearchLifecycleHarness({
|
||||
activeTools: ["init_experiment", "run_experiment", "log_experiment"],
|
||||
controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }],
|
||||
cwd: dir,
|
||||
});
|
||||
|
||||
await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx);
|
||||
const result = await harness.beforeAgentStartHandler?.({ systemPrompt: "BASE" }, harness.ctx);
|
||||
const systemPrompt =
|
||||
typeof result === "object" && result !== null && "systemPrompt" in result
|
||||
? String((result as { systemPrompt: string }).systemPrompt)
|
||||
: "";
|
||||
|
||||
expect(systemPrompt).toContain("### Local Playbook");
|
||||
expect(systemPrompt).toContain("### Current Segment Snapshot");
|
||||
expect(systemPrompt).toContain("### Pending Run");
|
||||
expect(systemPrompt).toContain("### Ideas backlog");
|
||||
expect(systemPrompt).toContain("Recent runs:");
|
||||
expect(systemPrompt).toContain("finish the `log_experiment` step before starting another benchmark");
|
||||
});
|
||||
});
|
||||
|
||||
describe("autoresearch lifecycle tool activation", () => {
|
||||
|
||||
@@ -245,6 +245,24 @@ describe("autoresearch tools", () => {
|
||||
});
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
runtime.lastRunChecks = { pass: true, output: "stale", duration: 1 };
|
||||
runtime.lastRunDuration = 1;
|
||||
runtime.lastRunAsi = { hypothesis: "stale" };
|
||||
runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "9999");
|
||||
runtime.lastRunNumber = 99;
|
||||
runtime.lastRunSummary = {
|
||||
checksDurationSeconds: 1,
|
||||
checksPass: true,
|
||||
checksTimedOut: false,
|
||||
command: "bash autoresearch.sh",
|
||||
durationSeconds: 1,
|
||||
parsedAsi: { hypothesis: "stale" },
|
||||
parsedMetrics: { runtime_ms: 10 },
|
||||
parsedPrimary: 10,
|
||||
passed: true,
|
||||
runDirectory: path.join(dir, ".autoresearch", "runs", "9999"),
|
||||
runNumber: 99,
|
||||
};
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
@@ -291,6 +309,12 @@ describe("autoresearch tools", () => {
|
||||
expect(configEntry.offLimits).toEqual(["src/generated"]);
|
||||
expect(configEntry.constraints).toEqual(["keep behavior stable"]);
|
||||
expect(configEntry.segmentFingerprint).toBe(createFingerprint(dir));
|
||||
expect(runtime.lastRunChecks).toBeNull();
|
||||
expect(runtime.lastRunDuration).toBeNull();
|
||||
expect(runtime.lastRunAsi).toBeNull();
|
||||
expect(runtime.lastRunArtifactDir).toBeNull();
|
||||
expect(runtime.lastRunNumber).toBeNull();
|
||||
expect(runtime.lastRunSummary).toBeNull();
|
||||
});
|
||||
|
||||
it("rejects init_experiment when the passed contract no longer matches autoresearch.md", async () => {
|
||||
@@ -349,6 +373,53 @@ describe("autoresearch tools", () => {
|
||||
expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects init_experiment while a previous run is still pending", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
writeAutoresearchWorkspace(dir);
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedPrimary: 10,
|
||||
runNumber: 1,
|
||||
}),
|
||||
);
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
const tool = createInitExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: {} as ExtensionAPI,
|
||||
});
|
||||
|
||||
const result = await tool.execute(
|
||||
"init-pending",
|
||||
{
|
||||
name: "Blocked",
|
||||
metric_name: "runtime_ms",
|
||||
metric_unit: "ms",
|
||||
direction: "lower",
|
||||
benchmark_command: "bash autoresearch.sh",
|
||||
scope_paths: ["src"],
|
||||
off_limits: [],
|
||||
constraints: [],
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createContext(dir),
|
||||
);
|
||||
|
||||
expect(result.content[0]).toEqual({
|
||||
type: "text",
|
||||
text: expect.stringContaining("has not been logged yet"),
|
||||
});
|
||||
expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false);
|
||||
});
|
||||
|
||||
it("refuses to start a new benchmark while a previous run artifact is still unlogged", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
@@ -512,6 +583,8 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-force-secondary-accept`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-keep`.cwd(dir).quiet();
|
||||
|
||||
fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 3;\n");
|
||||
fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n\n- focus on in-scope edits\n");
|
||||
@@ -618,6 +691,8 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-max-iterations-accept`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-force-secondary`.cwd(dir).quiet();
|
||||
|
||||
fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 99;\n");
|
||||
await Bun.write(
|
||||
@@ -707,6 +782,8 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-discard-cleanup`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-max-iterations`.cwd(dir).quiet();
|
||||
|
||||
fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 2;\n");
|
||||
await Bun.write(
|
||||
@@ -776,6 +853,7 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-discard`.cwd(dir).quiet();
|
||||
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0003", "run.json"),
|
||||
@@ -861,6 +939,280 @@ describe("autoresearch tools", () => {
|
||||
expect(runtime.state.results).toHaveLength(2);
|
||||
});
|
||||
|
||||
it("rejects log_experiment when configured secondary metrics are missing", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
writeAutoresearchWorkspace(dir);
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedMetrics: { memory_mb: 32, runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
runNumber: 1,
|
||||
}),
|
||||
);
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
runtime.state.metricName = "runtime_ms";
|
||||
runtime.state.metricUnit = "ms";
|
||||
runtime.state.secondaryMetrics = [
|
||||
{ name: "memory_mb", unit: "mb" },
|
||||
{ name: "tokens", unit: "" },
|
||||
];
|
||||
runtime.state.segmentFingerprint = createFingerprint(dir);
|
||||
runtime.lastRunSummary = {
|
||||
checksDurationSeconds: 0,
|
||||
checksPass: null,
|
||||
checksTimedOut: false,
|
||||
command: "bash autoresearch.sh",
|
||||
durationSeconds: 1,
|
||||
parsedAsi: null,
|
||||
parsedMetrics: { memory_mb: 32, runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
passed: true,
|
||||
runDirectory: path.join(dir, ".autoresearch", "runs", "0001"),
|
||||
runNumber: 1,
|
||||
};
|
||||
|
||||
const tool = createLogExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: {} as ExtensionAPI,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-missing-secondary",
|
||||
{
|
||||
commit: "initial",
|
||||
metric: 9,
|
||||
status: "discard",
|
||||
description: "missing tokens metric",
|
||||
metrics: { memory_mb: 32 },
|
||||
asi: {
|
||||
hypothesis: "watch memory only",
|
||||
rollback_reason: "missing required metrics",
|
||||
next_action_hint: "include all configured tradeoff metrics",
|
||||
},
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createContext(dir),
|
||||
);
|
||||
|
||||
expect(result.content[0]).toEqual({
|
||||
type: "text",
|
||||
text: expect.stringContaining("missing secondary metrics: tokens"),
|
||||
});
|
||||
expect(runtime.state.results).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("rejects new secondary metrics unless force is enabled", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
writeAutoresearchWorkspace(dir);
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
runNumber: 1,
|
||||
}),
|
||||
);
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
runtime.state.metricName = "runtime_ms";
|
||||
runtime.state.metricUnit = "ms";
|
||||
runtime.state.secondaryMetrics = [{ name: "memory_mb", unit: "mb" }];
|
||||
runtime.state.segmentFingerprint = createFingerprint(dir);
|
||||
runtime.lastRunSummary = {
|
||||
checksDurationSeconds: 0,
|
||||
checksPass: null,
|
||||
checksTimedOut: false,
|
||||
command: "bash autoresearch.sh",
|
||||
durationSeconds: 1,
|
||||
parsedAsi: null,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
passed: true,
|
||||
runDirectory: path.join(dir, ".autoresearch", "runs", "0001"),
|
||||
runNumber: 1,
|
||||
};
|
||||
|
||||
const tool = createLogExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: {} as ExtensionAPI,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-new-secondary",
|
||||
{
|
||||
commit: "initial",
|
||||
metric: 9,
|
||||
status: "discard",
|
||||
description: "introduce tokens metric",
|
||||
metrics: { memory_mb: 32, tokens: 100 },
|
||||
asi: {
|
||||
hypothesis: "watch an extra tradeoff",
|
||||
rollback_reason: "needs explicit opt-in",
|
||||
next_action_hint: "retry with force if the metric matters",
|
||||
},
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createContext(dir),
|
||||
);
|
||||
|
||||
expect(result.content[0]).toEqual({
|
||||
type: "text",
|
||||
text: expect.stringContaining("new secondary metrics require force=true: tokens"),
|
||||
});
|
||||
expect(runtime.state.results).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("accepts a new secondary metric when force is enabled", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
writeAutoresearchWorkspace(dir);
|
||||
|
||||
await $`git init`.cwd(dir).quiet();
|
||||
await $`git config user.email test@example.com`.cwd(dir).quiet();
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-force-secondary-accept`.cwd(dir).quiet();
|
||||
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
runNumber: 1,
|
||||
}),
|
||||
);
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
runtime.state.metricName = "runtime_ms";
|
||||
runtime.state.metricUnit = "ms";
|
||||
runtime.state.secondaryMetrics = [{ name: "memory_mb", unit: "mb" }];
|
||||
runtime.state.segmentFingerprint = createFingerprint(dir);
|
||||
runtime.lastRunSummary = {
|
||||
checksDurationSeconds: 0,
|
||||
checksPass: null,
|
||||
checksTimedOut: false,
|
||||
command: "bash autoresearch.sh",
|
||||
durationSeconds: 1,
|
||||
parsedAsi: null,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
passed: true,
|
||||
runDirectory: path.join(dir, ".autoresearch", "runs", "0001"),
|
||||
runNumber: 1,
|
||||
};
|
||||
|
||||
const tool = createLogExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: createGitApi(),
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-force-secondary",
|
||||
{
|
||||
commit: "initial",
|
||||
metric: 9,
|
||||
status: "discard",
|
||||
description: "force extra metric",
|
||||
force: true,
|
||||
metrics: { memory_mb: 32, tokens: 100 },
|
||||
asi: {
|
||||
hypothesis: "capture an extra tradeoff",
|
||||
rollback_reason: "benchmark was flat",
|
||||
next_action_hint: "keep collecting tokens when useful",
|
||||
},
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createContext(dir),
|
||||
);
|
||||
|
||||
expect(result.content[0]).toEqual({
|
||||
type: "text",
|
||||
text: expect.stringContaining("Logged run #1: discard"),
|
||||
});
|
||||
expect(runtime.state.secondaryMetrics).toContainEqual({ name: "tokens", unit: "" });
|
||||
});
|
||||
|
||||
it("rejects log_experiment at the tool boundary when asi is missing", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
writeAutoresearchWorkspace(dir);
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
JSON.stringify({
|
||||
command: "bash autoresearch.sh",
|
||||
completedAt: new Date().toISOString(),
|
||||
durationSeconds: 1,
|
||||
exitCode: 0,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
runNumber: 1,
|
||||
}),
|
||||
);
|
||||
|
||||
const runtime = createSessionRuntime();
|
||||
runtime.state.metricName = "runtime_ms";
|
||||
runtime.state.metricUnit = "ms";
|
||||
runtime.state.segmentFingerprint = createFingerprint(dir);
|
||||
runtime.lastRunSummary = {
|
||||
checksDurationSeconds: 0,
|
||||
checksPass: null,
|
||||
checksTimedOut: false,
|
||||
command: "bash autoresearch.sh",
|
||||
durationSeconds: 1,
|
||||
parsedAsi: null,
|
||||
parsedMetrics: { runtime_ms: 9 },
|
||||
parsedPrimary: 9,
|
||||
passed: true,
|
||||
runDirectory: path.join(dir, ".autoresearch", "runs", "0001"),
|
||||
runNumber: 1,
|
||||
};
|
||||
|
||||
const tool = createLogExperimentTool({
|
||||
dashboard: createDashboardStub(),
|
||||
getRuntime: () => runtime,
|
||||
pi: {} as ExtensionAPI,
|
||||
});
|
||||
const result = await tool.execute(
|
||||
"call-missing-asi",
|
||||
{
|
||||
commit: "initial",
|
||||
metric: 9,
|
||||
status: "keep",
|
||||
description: "missing asi",
|
||||
},
|
||||
undefined,
|
||||
undefined,
|
||||
createContext(dir),
|
||||
);
|
||||
|
||||
expect(result.content[0]).toEqual({
|
||||
type: "text",
|
||||
text: expect.stringContaining("asi is required"),
|
||||
});
|
||||
expect(runtime.state.results).toHaveLength(0);
|
||||
expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false);
|
||||
});
|
||||
|
||||
it("requires failed benchmarks to be logged as crash", async () => {
|
||||
const dir = makeTempDir();
|
||||
tempDirs.push(dir);
|
||||
@@ -1002,6 +1354,7 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-max-iterations-accept`.cwd(dir).quiet();
|
||||
|
||||
await Bun.write(
|
||||
path.join(dir, ".autoresearch", "runs", "0001", "run.json"),
|
||||
@@ -1164,6 +1517,7 @@ describe("autoresearch tools", () => {
|
||||
await $`git config user.name Test User`.cwd(dir).quiet();
|
||||
await $`git add .`.cwd(dir).quiet();
|
||||
await $`git commit -m initial`.cwd(dir).quiet();
|
||||
await $`git checkout -b autoresearch/test-discard-cleanup`.cwd(dir).quiet();
|
||||
|
||||
fs.mkdirSync(path.join(dir, "tmp-artifact"), { recursive: true });
|
||||
fs.writeFileSync(path.join(dir, "tmp-artifact", "result.txt"), "temporary benchmark output\n");
|
||||
|
||||
@@ -535,7 +535,9 @@ describe("applyHashlineEdits — heuristics", () => {
|
||||
];
|
||||
const result = applyHashlineEdits(content, edits);
|
||||
expect(result.lines).toBe("if (ok) {\n runSafe();\n}\n}\nafter();");
|
||||
expect(result.warnings).toBeUndefined();
|
||||
expect(result.warnings).toHaveLength(1);
|
||||
expect(result.warnings?.[0]).toContain("Possible boundary duplication");
|
||||
expect(result.warnings?.[0]).toContain("set `end` to 3#RZ");
|
||||
});
|
||||
|
||||
it("preserves duplicated trailing content when replacement re-emits the next line", () => {
|
||||
@@ -550,7 +552,9 @@ describe("applyHashlineEdits — heuristics", () => {
|
||||
];
|
||||
const result = applyHashlineEdits(content, edits);
|
||||
expect(result.lines).toBe("start\n newCall();\nnextCall();\nnextCall();\nafter();");
|
||||
expect(result.warnings).toBeUndefined();
|
||||
expect(result.warnings).toHaveLength(1);
|
||||
expect(result.warnings?.[0]).toContain("Possible boundary duplication");
|
||||
expect(result.warnings?.[0]).toContain("set `end` to 3#HR");
|
||||
});
|
||||
|
||||
it("preserves duplicated leading content when replacement re-emits the previous line", () => {
|
||||
|
||||
Reference in New Issue
Block a user