From 855d89cc5eab2607428293926adead3bc438f4aa Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 01:53:06 +0100 Subject: [PATCH 01/22] feat(coding-agent): added inline markdown rendering with theme-aware styling - Added renderInlineMarkdown() utility function to support inline markdown rendering with optional base color styling. - Refactored ask tool to render questions and option labels with markdown formatting for improved text styling. - Updated hook-input and hook-selector components to render titles as markdown with theme-aware styling. - Implemented recursive token processing for nested markdown elements including bold, italic, code, links, and strikethrough. Fixes #491 --- packages/coding-agent/CHANGELOG.md | 3 + .../src/modes/components/hook-input.ts | 12 +-- .../src/modes/components/hook-selector.ts | 22 +++-- packages/coding-agent/src/tools/ask.ts | 86 ++++++++++++------- packages/tui/CHANGELOG.md | 5 +- packages/tui/src/components/markdown.ts | 67 ++++++++++++++- packages/tui/test/markdown.test.ts | 11 ++- 7 files changed, 158 insertions(+), 48 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d52f924bc..955fe5505 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Added - Added ACP (Agent Client Protocol) mode for headless agent operation via `--mode acp` @@ -9,6 +10,8 @@ ### Changed +- Updated ask tool rendering to support markdown formatting in questions and option labels +- Refactored hook input and selector components to render titles as markdown for richer text formatting - Changed session collection to include sessions with zero messages, enabling ACP mode to create discoverable sessions immediately - Changed session persistence logic to use atomic file rewrite when flushing unflushed sessions to prevent duplication diff --git a/packages/coding-agent/src/modes/components/hook-input.ts b/packages/coding-agent/src/modes/components/hook-input.ts index 06a192457..eb4d01f66 100644 --- a/packages/coding-agent/src/modes/components/hook-input.ts +++ b/packages/coding-agent/src/modes/components/hook-input.ts @@ -1,8 +1,8 @@ /** * Simple text input component for hooks. */ -import { Container, Input, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; -import { theme } from "../../modes/theme/theme"; +import { Container, Input, Markdown, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; +import { getMarkdownTheme, theme } from "../../modes/theme/theme"; import { CountdownTimer } from "./countdown-timer"; import { DynamicBorder } from "./dynamic-border"; @@ -16,7 +16,7 @@ export class HookInputComponent extends Container { #input: Input; #onSubmitCallback: (value: string) => void; #onCancelCallback: () => void; - #titleText: Text; + #titleComponent: Markdown; #baseTitle: string; #countdown: CountdownTimer | undefined; @@ -36,15 +36,15 @@ export class HookInputComponent extends Container { this.addChild(new DynamicBorder()); this.addChild(new Spacer(1)); - this.#titleText = new Text(theme.fg("accent", title), 1, 0); - this.addChild(this.#titleText); + this.#titleComponent = new Markdown(title, 1, 0, getMarkdownTheme(), { color: t => theme.fg("accent", t) }); + this.addChild(this.#titleComponent); this.addChild(new Spacer(1)); if (opts?.timeout && opts.timeout > 0 && opts.tui) { this.#countdown = new CountdownTimer( opts.timeout, opts.tui, - s => this.#titleText.setText(theme.fg("accent", `${this.#baseTitle} (${s}s)`)), + s => this.#titleComponent.setText(`${this.#baseTitle} (${s}s)`), () => { opts.onTimeout?.(); this.#onCancelCallback(); diff --git a/packages/coding-agent/src/modes/components/hook-selector.ts b/packages/coding-agent/src/modes/components/hook-selector.ts index 5c2dec8ec..c00aea1d1 100644 --- a/packages/coding-agent/src/modes/components/hook-selector.ts +++ b/packages/coding-agent/src/modes/components/hook-selector.ts @@ -4,8 +4,10 @@ */ import { Container, + Markdown, matchesKey, padding, + renderInlineMarkdown, replaceTabs, Spacer, Text, @@ -13,7 +15,7 @@ import { truncateToWidth, visibleWidth, } from "@oh-my-pi/pi-tui"; -import { theme } from "../../modes/theme/theme"; +import { getMarkdownTheme, theme } from "../../modes/theme/theme"; import { CountdownTimer } from "./countdown-timer"; import { DynamicBorder } from "./dynamic-border"; @@ -59,7 +61,7 @@ export class HookSelectorComponent extends Container { #outlinedList: OutlinedList | undefined; #onSelectCallback: (option: string) => void; #onCancelCallback: () => void; - #titleText: Text; + #titleComponent: Markdown; #baseTitle: string; #countdown: CountdownTimer | undefined; #onLeftCallback: (() => void) | undefined; @@ -85,15 +87,15 @@ export class HookSelectorComponent extends Container { this.addChild(new DynamicBorder()); this.addChild(new Spacer(1)); - this.#titleText = new Text(theme.fg("accent", title), 1, 0); - this.addChild(this.#titleText); + this.#titleComponent = new Markdown(title, 1, 0, getMarkdownTheme(), { color: t => theme.fg("accent", t) }); + this.addChild(this.#titleComponent); this.addChild(new Spacer(1)); if (opts?.timeout && opts.timeout > 0 && opts.tui) { this.#countdown = new CountdownTimer( opts.timeout, opts.tui, - s => this.#titleText.setText(theme.fg("accent", `${this.#baseTitle} (${s}s)`)), + s => this.#titleComponent.setText(`${this.#baseTitle} (${s}s)`), () => { opts?.onTimeout?.(); // Auto-select current option on timeout (typically the first/recommended option) @@ -131,12 +133,14 @@ export class HookSelectorComponent extends Container { ); const endIndex = Math.min(startIndex + this.#maxVisible, this.#options.length); + const mdTheme = getMarkdownTheme(); for (let i = startIndex; i < endIndex; i++) { const isSelected = i === this.#selectedIndex; - const text = isSelected - ? theme.fg("accent", `${theme.nav.cursor} `) + theme.fg("accent", this.#options[i]) - : ` ${theme.fg("text", this.#options[i])}`; - lines.push(text); + const label = isSelected + ? renderInlineMarkdown(this.#options[i], mdTheme, t => theme.fg("accent", t)) + : renderInlineMarkdown(this.#options[i], mdTheme, t => theme.fg("text", t)); + const prefix = isSelected ? theme.fg("accent", `${theme.nav.cursor} `) : " "; + lines.push(prefix + label); } if (startIndex > 0 || endIndex < this.#options.length) { diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index f611bba7c..c2f2fc7f1 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -16,13 +16,12 @@ */ import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; -import type { Component } from "@oh-my-pi/pi-tui"; -import { TERMINAL, Text } from "@oh-my-pi/pi-tui"; +import { type Component, Container, Markdown, renderInlineMarkdown, TERMINAL, Text } from "@oh-my-pi/pi-tui"; import { untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; import { renderPromptTemplate } from "../config/prompt-templates"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { type Theme, theme } from "../modes/theme/theme"; +import { getMarkdownTheme, type Theme, theme } from "../modes/theme/theme"; import askDescription from "../prompts/tools/ask.md" with { type: "text" }; import { renderStatusLine } from "../tui"; import type { ToolSession } from "."; @@ -574,10 +573,13 @@ interface AskRenderArgs { export const askToolRenderer = { renderCall(args: AskRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component { const label = formatTitle("Ask", uiTheme); + const mdTheme = getMarkdownTheme(); + const accentStyle = { color: (t: string) => uiTheme.fg("accent", t) }; // Multi-part questions if (args.questions && args.questions.length > 0) { - let text = `${label} ${uiTheme.fg("muted", `${args.questions.length} questions`)}`; + const container = new Container(); + container.addChild(new Text(`${label} ${uiTheme.fg("muted", `${args.questions.length} questions`)}`, 0, 0)); for (let i = 0; i < args.questions.length; i++) { const q = args.questions[i]; @@ -585,25 +587,29 @@ export const askToolRenderer = { const qBranch = isLastQ ? uiTheme.tree.last : uiTheme.tree.branch; const continuation = isLastQ ? " " : uiTheme.tree.vertical; - // Question line with metadata const meta: string[] = []; if (q.multi) meta.push("multi"); if (q.options?.length) meta.push(`options:${q.options.length}`); const metaStr = meta.length > 0 ? uiTheme.fg("dim", ` · ${meta.join(" · ")}`) : ""; - text += `\n ${uiTheme.fg("dim", qBranch)} ${uiTheme.fg("dim", `[${q.id}]`)} ${uiTheme.fg("accent", q.question)}${metaStr}`; + container.addChild( + new Text(` ${uiTheme.fg("dim", qBranch)} ${uiTheme.fg("dim", `[${q.id}]`)}${metaStr}`, 0, 0), + ); + container.addChild(new Markdown(q.question, 3, 0, mdTheme, accentStyle)); - // Options under question if (q.options?.length) { + let optText = ""; for (let j = 0; j < q.options.length; j++) { const opt = q.options[j]; const isLastOpt = j === q.options.length - 1; const optBranch = isLastOpt ? uiTheme.tree.last : uiTheme.tree.branch; - text += `\n ${uiTheme.fg("dim", continuation)} ${uiTheme.fg("dim", optBranch)} ${uiTheme.fg("dim", uiTheme.checkbox.unchecked)} ${uiTheme.fg("muted", opt.label)}`; + const optLabel = renderInlineMarkdown(opt.label, mdTheme, t => uiTheme.fg("muted", t)); + optText += `\n ${uiTheme.fg("dim", continuation)} ${uiTheme.fg("dim", optBranch)} ${uiTheme.fg("dim", uiTheme.checkbox.unchecked)} ${optLabel}`; } + container.addChild(new Text(optText, 0, 0)); } } - return new Text(text, 0, 0); + return container; } // Single question @@ -611,22 +617,26 @@ export const askToolRenderer = { return new Text(formatErrorMessage("No question provided", uiTheme), 0, 0); } - let text = `${label} ${uiTheme.fg("accent", args.question)}`; + const container = new Container(); const meta: string[] = []; if (args.multi) meta.push("multi"); if (args.options?.length) meta.push(`options:${args.options.length}`); - text += formatMeta(meta, uiTheme); + container.addChild(new Text(`${label}${formatMeta(meta, uiTheme)}`, 0, 0)); + container.addChild(new Markdown(args.question, 1, 0, mdTheme, accentStyle)); if (args.options?.length) { + let optText = ""; for (let i = 0; i < args.options.length; i++) { const opt = args.options[i]; const isLast = i === args.options.length - 1; const branch = isLast ? uiTheme.tree.last : uiTheme.tree.branch; - text += `\n ${uiTheme.fg("dim", branch)} ${uiTheme.fg("dim", uiTheme.checkbox.unchecked)} ${uiTheme.fg("muted", opt.label)}`; + const optLabel = renderInlineMarkdown(opt.label, mdTheme, t => uiTheme.fg("muted", t)); + optText += `\n ${uiTheme.fg("dim", branch)} ${uiTheme.fg("dim", uiTheme.checkbox.unchecked)} ${optLabel}`; } + container.addChild(new Text(optText, 0, 0)); } - return new Text(text, 0, 0); + return container; }, renderResult( @@ -635,6 +645,9 @@ export const askToolRenderer = { uiTheme: Theme, ): Component { const { details } = result; + const mdTheme = getMarkdownTheme(); + const accentStyle = { color: (t: string) => uiTheme.fg("accent", t) }; + if (!details) { const txt = result.content[0]; const fallback = txt?.type === "text" && txt.text ? txt.text : ""; @@ -655,7 +668,8 @@ export const askToolRenderer = { }, uiTheme, ); - let text = header; + const container = new Container(); + container.addChild(new Text(header, 0, 0)); for (let i = 0; i < details.results.length; i++) { const r = details.results[i]; @@ -667,22 +681,31 @@ export const askToolRenderer = { ? uiTheme.styledSymbol("status.success", "success") : uiTheme.styledSymbol("status.warning", "warning"); - text += `\n ${uiTheme.fg("dim", branch)} ${statusIcon} ${uiTheme.fg("dim", `[${r.id}]`)} ${uiTheme.fg("accent", r.question)}`; + container.addChild( + new Text(` ${uiTheme.fg("dim", branch)} ${statusIcon} ${uiTheme.fg("dim", `[${r.id}]`)}`, 0, 0), + ); + container.addChild(new Markdown(r.question, 3, 0, mdTheme, accentStyle)); + let answerText = ""; if (r.customInput) { - text += `\n${continuation}${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.success", "success")} ${uiTheme.fg("toolOutput", r.customInput)}`; + answerText = `${continuation}${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.success", "success")} ${uiTheme.fg("toolOutput", r.customInput)}`; } else if (r.selectedOptions.length > 0) { for (let j = 0; j < r.selectedOptions.length; j++) { const isLast = j === r.selectedOptions.length - 1; const optBranch = isLast ? uiTheme.tree.last : uiTheme.tree.branch; - text += `\n${continuation}${uiTheme.fg("dim", optBranch)} ${uiTheme.fg("success", uiTheme.checkbox.checked)} ${uiTheme.fg("toolOutput", r.selectedOptions[j])}`; + const selectedLabel = renderInlineMarkdown(r.selectedOptions[j], mdTheme, t => + uiTheme.fg("toolOutput", t), + ); + answerText += `\n${continuation}${uiTheme.fg("dim", optBranch)} ${uiTheme.fg("success", uiTheme.checkbox.checked)} ${selectedLabel}`; } } else { - text += `\n${continuation}${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.warning", "warning")} ${uiTheme.fg("warning", "Cancelled")}`; + answerText = `${continuation}${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.warning", "warning")} ${uiTheme.fg("warning", "Cancelled")}`; + } + if (answerText) { + container.addChild(new Text(answerText, 0, 0)); } } - - return new Text(text, 0, 0); + return container; } // Single question result @@ -693,25 +716,28 @@ export const askToolRenderer = { } const hasSelection = details.customInput || (details.selectedOptions && details.selectedOptions.length > 0); - const header = renderStatusLine( - { icon: hasSelection ? "success" : "warning", title: "Ask", description: details.question }, - uiTheme, - ); - - let text = header; + const header = renderStatusLine({ icon: hasSelection ? "success" : "warning", title: "Ask" }, uiTheme); + const container = new Container(); + container.addChild(new Text(header, 0, 0)); + container.addChild(new Markdown(details.question, 1, 0, mdTheme, accentStyle)); + let answerText = ""; if (details.customInput) { - text += `\n ${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.success", "success")} ${uiTheme.fg("toolOutput", details.customInput)}`; + answerText = ` ${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.success", "success")} ${uiTheme.fg("toolOutput", details.customInput)}`; } else if (details.selectedOptions && details.selectedOptions.length > 0) { for (let i = 0; i < details.selectedOptions.length; i++) { const isLast = i === details.selectedOptions.length - 1; const branch = isLast ? uiTheme.tree.last : uiTheme.tree.branch; - text += `\n ${uiTheme.fg("dim", branch)} ${uiTheme.fg("success", uiTheme.checkbox.checked)} ${uiTheme.fg("toolOutput", details.selectedOptions[i])}`; + const selectedLabel = renderInlineMarkdown(details.selectedOptions[i], mdTheme, t => + uiTheme.fg("toolOutput", t), + ); + answerText += `\n ${uiTheme.fg("dim", branch)} ${uiTheme.fg("success", uiTheme.checkbox.checked)} ${selectedLabel}`; } } else { - text += `\n ${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.warning", "warning")} ${uiTheme.fg("warning", "Cancelled")}`; + answerText = ` ${uiTheme.fg("dim", uiTheme.tree.last)} ${uiTheme.styledSymbol("status.warning", "warning")} ${uiTheme.fg("warning", "Cancelled")}`; } + container.addChild(new Text(answerText, 0, 0)); - return new Text(text, 0, 0); + return container; }, }; diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 45a47cc58..d13917853 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Added + +- Added `renderInlineMarkdown()` function to render inline markdown (bold, italic, code, links, strikethrough) to styled strings ## [13.14.1] - 2026-03-21 ### Added @@ -652,4 +655,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon ### Fixed -- **Readline-style Ctrl+W**: Now skips trailing whitespace before deleting the preceding word, matching standard readline behavior. ([#306](https://github.com/badlogic/pi-mono/pull/306) by [@kim0](https://github.com/kim0)) +- **Readline-style Ctrl+W**: Now skips trailing whitespace before deleting the preceding word, matching standard readline behavior. ([#306](https://github.com/badlogic/pi-mono/pull/306) by [@kim0](https://github.com/kim0)) \ No newline at end of file diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index 557cd27f2..2a8a32ee3 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -1,4 +1,4 @@ -import { marked, type Token } from "marked"; +import { marked, type Token, type Tokens } from "marked"; import type { SymbolTheme } from "../symbols"; import { TERMINAL } from "../terminal-capabilities"; import type { Component } from "../tui"; @@ -838,3 +838,68 @@ export class Markdown implements Component { return lines; } } + +/** + * Render inline markdown (bold, italic, code, links, strikethrough) to a styled string. + * Unlike the full Markdown component, this produces a single line with no block-level elements. + */ +export function renderInlineMarkdown(text: string, mdTheme: MarkdownTheme, baseColor?: (t: string) => string): string { + const tokens = marked.lexer(text); + const applyText = baseColor ?? ((t: string) => t); + let result = ""; + for (const token of tokens) { + if (token.type === "paragraph" && token.tokens) { + result += renderInlineTokens(token.tokens, mdTheme, applyText); + } else if (token.type === "list") { + result += token.items + .map((item: Tokens.ListItem, index: number) => { + const prefix = token.ordered ? `${(token.start || 1) + index}. ` : "• "; + const content = item.tokens ? renderInlineTokens(item.tokens, mdTheme, applyText) : applyText(item.text); + return `${applyText(prefix)}${content}`; + }) + .join(applyText(" ")); + } else if ("text" in token && typeof token.text === "string") { + result += applyText(token.text); + } + } + return result; +} + +function renderInlineTokens(tokens: Token[], mdTheme: MarkdownTheme, applyText: (t: string) => string): string { + let result = ""; + const styleReset = applyText(""); + for (const token of tokens) { + switch (token.type) { + case "text": + if (token.tokens && token.tokens.length > 0) { + result += renderInlineTokens(token.tokens, mdTheme, applyText); + } else { + result += applyText(token.text); + } + break; + case "strong": + result += mdTheme.bold(renderInlineTokens(token.tokens || [], mdTheme, applyText)) + styleReset; + break; + case "em": + result += mdTheme.italic(renderInlineTokens(token.tokens || [], mdTheme, applyText)) + styleReset; + break; + case "codespan": + result += mdTheme.code(token.text) + styleReset; + break; + case "del": + result += mdTheme.strikethrough(renderInlineTokens(token.tokens || [], mdTheme, applyText)) + styleReset; + break; + case "link": { + const linkText = renderInlineTokens(token.tokens || [], mdTheme, applyText); + result += mdTheme.link(mdTheme.underline(linkText)) + styleReset; + break; + } + default: + if ("text" in token && typeof token.text === "string") { + result += applyText(token.text); + } + break; + } + } + return result; +} diff --git a/packages/tui/test/markdown.test.ts b/packages/tui/test/markdown.test.ts index 9eb1830c5..58dc63616 100644 --- a/packages/tui/test/markdown.test.ts +++ b/packages/tui/test/markdown.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from "bun:test"; import type { Terminal as XtermTerminalType } from "@xterm/headless"; import { Chalk } from "chalk"; -import { Markdown } from "../src/components/markdown.js"; +import { Markdown, renderInlineMarkdown } from "../src/components/markdown.js"; import { type Component, TUI } from "../src/tui.js"; import { defaultMarkdownTheme } from "./test-themes.js"; import { VirtualTerminal } from "./virtual-terminal.js"; @@ -19,6 +19,15 @@ function getCellItalic(terminal: VirtualTerminal, row: number, col: number): num return cell!.isItalic(); } +describe("renderInlineMarkdown", () => { + it("preserves ordered list items as visible inline text", () => { + const rendered = renderInlineMarkdown("1. Review against a base branch (PR Style)", defaultMarkdownTheme); + const plain = rendered.replace(/\x1b\[[0-9;]*m/g, ""); + + expect(plain).toBe("1. Review against a base branch (PR Style)"); + }); +}); + describe("Markdown component", () => { describe("Nested lists", () => { it("should render simple nested list", () => { From d4b6e869cb66fc423dae272eaa6cb8861451ed54 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 12:38:33 +0100 Subject: [PATCH 02/22] feat(coding-agent): implemented pattern-based bash interceptor rules + exclusive range semantics - Changed bash interceptor configuration from boolean flags to customizable pattern-based rules array. - Clarified hashline range replace semantics: end parameter is now strictly exclusive boundary. - Fixed bash interceptor to apply built-in default rules when no custom patterns are configured. - Updated hashline range validation and calculations to enforce exclusive end semantics throughout. --- packages/coding-agent/CHANGELOG.md | 7 +- .../src/config/settings-schema.ts | 48 +++++++--- packages/coding-agent/src/config/settings.ts | 5 +- packages/coding-agent/src/patch/hashline.ts | 40 ++++++--- .../src/prompts/tools/hashline.md | 54 +++++------- .../src/tools/bash-interceptor.ts | 40 +-------- .../coding-agent/test/core/hashline.test.ts | 75 ++++++++++++---- .../session-selector-delete.test.ts | 20 ++--- packages/coding-agent/test/tools.test.ts | 87 ++++++++++++++++++- 9 files changed, 243 insertions(+), 133 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 955fe5505..704a1efde 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Added - Added ACP (Agent Client Protocol) mode for headless agent operation via `--mode acp` @@ -10,11 +9,17 @@ ### Changed +- Updated bash interceptor configuration to use customizable pattern rules instead of individual boolean flags +- Clarified hashline range replace semantics: `end` parameter is now strictly exclusive (the line it points to survives and is not consumed) - Updated ask tool rendering to support markdown formatting in questions and option labels - Refactored hook input and selector components to render titles as markdown for richer text formatting - Changed session collection to include sessions with zero messages, enabling ACP mode to create discoverable sessions immediately - Changed session persistence logic to use atomic file rewrite when flushing unflushed sessions to prevent duplication +### Fixed + +- Fixed bash interceptor to apply built-in default rules when no custom patterns are configured + ## [13.14.0] - 2026-03-20 ### Added diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 09ceb4274..260855051 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -139,6 +139,43 @@ type SettingDef = // under `as const` while still letting SettingValue infer the correct element type. const EMPTY_STRING_ARRAY: string[] = []; const EMPTY_STRING_RECORD: Record = {}; +export const DEFAULT_BASH_INTERCEPTOR_RULES: BashInterceptorRule[] = [ + { + pattern: "^\\s*(cat|head|tail|less|more)\\s+", + tool: "read", + message: "Use the `read` tool instead of cat/head/tail. It provides better context and handles binary files.", + }, + { + pattern: "^\\s*(grep|rg|ripgrep|ag|ack)\\s+", + tool: "grep", + message: "Use the `grep` tool instead of grep/rg. It respects .gitignore and provides structured output.", + }, + { + pattern: "^\\s*(find|fd|locate)\\s+.*(-name|-iname|-type|--type|-glob)", + tool: "find", + message: "Use the `find` tool instead of find/fd. It respects .gitignore and is faster for glob patterns.", + }, + { + pattern: "^\\s*sed\\s+(-i|--in-place)", + tool: "edit", + message: "Use the `edit` tool instead of sed -i. It provides diff preview and fuzzy matching.", + }, + { + pattern: "^\\s*perl\\s+.*-[pn]?i", + tool: "edit", + message: "Use the `edit` tool instead of perl -i. It provides diff preview and fuzzy matching.", + }, + { + pattern: "^\\s*awk\\s+.*-i\\s+inplace", + tool: "edit", + message: "Use the `edit` tool instead of awk -i inplace. It provides diff preview and fuzzy matching.", + }, + { + pattern: "^\\s*(echo|printf|cat\\s*<<)\\s+.*[^|]>\\s*\\S", + tool: "write", + message: "Use the `write` tool instead of echo/cat redirection. It handles encoding and provides confirmation.", + }, +]; export const SETTINGS_SCHEMA = { // ──────────────────────────────────────────────────────────────────────── @@ -943,16 +980,7 @@ export const SETTINGS_SCHEMA = { default: false, ui: { tab: "editing", label: "Bash Interceptor", description: "Block shell commands that have dedicated tools" }, }, - - "bashInterceptor.simpleLs": { - type: "boolean", - default: true, - ui: { - tab: "editing", - label: "Intercept `ls`", - description: "Intercept bare ls commands (when interceptor is enabled)", - }, - }, + "bashInterceptor.patterns": { type: "array", default: DEFAULT_BASH_INTERCEPTOR_RULES }, // Python "python.toolMode": { diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 1ecb255c3..3740ecc23 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -341,10 +341,7 @@ export class Settings { * Get bash interceptor rules (typed accessor for complex array config). */ getBashInterceptorRules(): BashInterceptorRule[] { - const patterns = (this.#merged.bashInterceptor as { patterns?: unknown[] })?.patterns; - if (!Array.isArray(patterns)) return []; - - return patterns.filter((p): p is BashInterceptorRule => typeof p === "object" && p !== null && "pattern" in p); + return this.get("bashInterceptor.patterns"); } /** diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 8bd67ddfe..493770dbc 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -15,6 +15,13 @@ import type { HashMismatch } from "./types"; export type Anchor = { line: number; hash: string }; +/** + * Edit operation on hashline-addressed content. + * + * For range replace: `pos` is inclusive (first consumed line), + * `end` is **exclusive** (first surviving line after the range). + * The consumed range is `[pos.line, end.line - 1]`. + */ export type HashlineEdit = | { op: "replace"; pos: Anchor; end?: Anchor; lines: string[] } | { op: "append"; pos?: Anchor; lines: string[] } @@ -470,9 +477,8 @@ function shouldAutocorrect(line: string, otherLine: string): boolean { /** * Apply an array of hashline edits to file content. * - * Each edit operation identifies target lines directly (`replace`, - * `append`, `prepend`). Line references are resolved via {@link parseTag} - * and hashes validated before any mutation. + * For range replace, `end` is **exclusive**: the consumed range is + * `[pos.line, end.line - 1]` and the line at `end` survives. * * Edits are sorted bottom-up (highest effective line first) so earlier * splices don't invalidate later line numbers. @@ -518,8 +524,10 @@ export function applyHashlineEdits( const startValid = validateRef(edit.pos); const endValid = validateRef(edit.end); if (!startValid || !endValid) continue; - if (edit.pos.line > edit.end.line) { - throw new Error(`Range start line ${edit.pos.line} must be <= end line ${edit.end.line}`); + if (edit.pos.line >= edit.end.line) { + throw new Error( + `Range start line ${edit.pos.line} must be < end line ${edit.end.line} (end is exclusive)`, + ); } } else { if (!validateRef(edit.pos)) continue; @@ -597,10 +605,13 @@ export function applyHashlineEdits( case "replace": if (!edit.end) { sortLine = edit.pos.line; + precedence = 0; } else { sortLine = edit.end.line; + // Range replaces must run after edits anchored on the surviving end line, + // so those line-number references still point at the same survivor. + precedence = 3; } - precedence = 0; break; case "append": sortLine = edit.pos ? edit.pos.line : fileLines.length + 1; @@ -634,19 +645,22 @@ export function applyHashlineEdits( fileLines.splice(edit.pos.line - 1, 1, ...newLines); trackFirstChanged(edit.pos.line); } else { - const count = edit.end.line - edit.pos.line + 1; + // end is exclusive: consumed range is [pos.line, end.line - 1] + const count = edit.end.line - edit.pos.line; const newLines = [...edit.lines]; + // The end line itself survives (exclusive). If the model re-emits it + // in lines, that's a duplication mistake — auto-correct by popping. const trailingReplacementLine = newLines[newLines.length - 1]?.trimEnd(); - const nextSurvivingLine = fileLines[edit.end.line]?.trimEnd(); + const nextSurvivingLine = fileLines[edit.end.line - 1]?.trimEnd(); if ( shouldAutocorrect(trailingReplacementLine, nextSurvivingLine) && - // Safety: only correct when end-line content differs from the duplicate. - // If end already points to the boundary, matching next line is coincidence. - fileLines[edit.end.line - 1]?.trimEnd() !== trailingReplacementLine + // Safety: only correct when the last consumed line differs from the duplicate. + // If the last consumed line is the same as the surviving end line, it's coincidence. + fileLines[edit.end.line - 2]?.trimEnd() !== trailingReplacementLine ) { newLines.pop(); warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed trailing replacement line "${trailingReplacementLine}" that duplicated next surviving line`, + `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}..${edit.end.line}#${edit.end.hash}: removed trailing replacement line "${trailingReplacementLine}" that duplicated the surviving end line`, ); } const leadingReplacementLine = newLines[0]?.trimEnd(); @@ -659,7 +673,7 @@ export function applyHashlineEdits( ) { newLines.shift(); warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed leading replacement line "${leadingReplacementLine}" that duplicated preceding surviving line`, + `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}..${edit.end.line}#${edit.end.hash}: removed leading replacement line "${leadingReplacementLine}" that duplicated preceding surviving line`, ); } fileLines.splice(edit.pos.line - 1, count, ...newLines); diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index 2b7fea2e3..821564d68 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -11,14 +11,13 @@ Read the file first to get fresh tags. Submit one `edit` call per file with all - if `replace`: first line to rewrite - if `prepend`: line to insert new lines **before**; omit for beginning of file - if `append`: line to insert new lines **after**; omit for end of file -**`edits[n].end`** — range replace only. The last line of the range (inclusive). Omit for single-line replace. +**`edits[n].end`** — range replace only. The first line **after** the range (exclusive — this line survives). Omit for single-line replace. **`edits[n].lines`** — the replacement content: - - for `replace`: the exact lines that will replace `[pos, end??pos]` inclusively (or the single `pos` line when `end` is omitted) + - for `replace`: the lines that will replace `[pos, end)`. Everything from `pos` up to (but not including) `end` is removed; `lines` is inserted in its place. - for `prepend`/`append`: the new lines to insert - `[""]` — blank line - `null` or `[]` — delete if replace -- If `lines` contains content that already exists after `end`, those lines **will be duplicated** in the output. -- Keep `lines` to exactly what belongs inside the consumed range. +- **`end` is exclusive — the line it points to stays in the file.** You do not need to re-emit it in `lines`. If you accidentally include it in `lines`, it will be duplicated. - Ops are applied bottom-up. Tags **MUST** be referenced from the most recent `read` output. @@ -71,22 +70,22 @@ Single line — `lines: null` deletes entirely: }] } ``` -Range — remove the legacy block (lines 10–11): +Range — remove the legacy block (lines 10–11). `end` points to line 12 (the line after the range): ``` { path: "util.ts", edits: [{ op: "replace", pos: {{hlineref 10 "\t// TODO: remove after migration"}}, - end: {{hlineref 11 "\tlegacy();"}}, + end: {{hlineref 12 "\ttry {"}}, lines: null }] } ``` - -Replace the catch body with smarter error handling. Shape (a): `pos` is the first body line, `end` is the last body line. The catch header (line 14) and its closer (line 17) are outside the range and stay untouched. + +Replace the catch body with smarter error handling. `pos` is the first body line, `end` is the closer — the closer survives automatically. When changing body content, replace the **entire** body span — not just one line inside it. Patching one line leaves the rest of the body stale. ``` @@ -95,7 +94,7 @@ When changing body content, replace the **entire** body span — not just one li edits: [{ op: "replace", pos: {{hlineref 15 "\t\tconsole.error(err);"}}, - end: {{hlineref 16 "\t\treturn null;"}}, + end: {{hlineref 17 "\t}"}}, lines: [ "\t\tif (isEnoent(err)) return null;", "\t\tthrow err;" @@ -103,28 +102,15 @@ When changing body content, replace the **entire** body span — not just one li }] } ``` +Result: lines 15–16 are replaced. The `\t}` on line 17 stays because `end` is exclusive. - -Simplify `beta()` to a one-liner. Shape (b): `pos`=header, `end`=closer, re-emit all in `lines`. + +Simplify `beta()` to a one-liner. `pos`=header (consumed), `end`=the line **after** the block (survives). -Bad — `end` stops at the inner `\t}` on line 17, so the outer `}` on line 18 survives. Result: two consecutive `}` lines. -``` -{ - path: "util.ts", - edits: [{ - op: "replace", - pos: {{hlineref 9 "function beta() {"}}, - end: {{hlineref 17 "\t}"}}, - lines: [ - "function beta() {", - "\treturn parse(data);", - "}" - ] - }] -} -``` -Good — `end` includes the function's own `}` on line 18, so the old closer is consumed: +Since `}` on line 18 is the last line of `beta()` and we want to consume it, `end` must point to the next line after the block. When line 18 is the last line of the file, omit `end` — single-line replace plus a delete of lines 10–17 first, or use `write` to rewrite the file. + +When there IS a line after the block: ``` { path: "util.ts", @@ -134,12 +120,12 @@ Good — `end` includes the function's own `}` on line 18, so the old closer is end: {{hlineref 18 "}"}}, lines: [ "function beta() {", - "\treturn parse(data);", - "}" + "\treturn parse(data);" ] }] } ``` +Result: lines 9–17 are consumed (replaced). `}` on line 18 survives as beta's closer — no need to re-emit it. @@ -149,7 +135,7 @@ Bad — if you need to change code on both sides of that line, replacing just th Good — choose one of two safe shapes instead: - move inward and replace only body-owned lines -- expand outward and replace one whole owned block, consuming its real closer/separator too +- expand outward and replace one whole owned block @@ -180,9 +166,9 @@ Use a trailing `""` to preserve the blank line between sibling declarations. - For `append`/`prepend`, `lines` **MUST** contain only the newly introduced content. Do not re-emit surrounding content, or terminators that already exist. - When changing existing code near a block tail or closing delimiter, default to `replace` over the owned span instead of inserting around the boundary. - When adding a sibling declaration, default to `prepend` on the next sibling declaration instead of `append` on the previous block's closing brace. -- **Block boundaries travel together.** For a block `{ header / body / closer }`, there are exactly two valid replace shapes: (a) replace only the body — `pos`=first body line, `end`=last body line, leave the header and closer untouched; or (b) replace the whole block — `pos`=header, `end`=closer, re-emit all three in `lines`. Never split them: do not set `end` to the closer while omitting it from `lines` (deletes it), and do not emit the closer in `lines` without including it in `end` (duplicates it). This applies to every block terminator: `}`, `continue`, `break`, `return`, `throw`. -- **Never target shared boundary lines.** Do not use `replace` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct including its true trailing delimiter. -- **`lines` must not extend past `end`.** `lines` replaces exactly `pos..end`. Content after `end` survives. If you include lines in `lines` that exist after `end`, they will appear twice. Either extend `end` to cover all lines you are re-emitting, or remove the extra lines from `lines`. +- **`end` is the boundary you want to keep.** Point `end` at the closing delimiter (`}`, `)`, ``) when you want it to survive. Point `end` past it when you want to consume it. Do **not** include the `end` line in `lines` — it survives on its own. +- **Never target shared boundary lines.** Do not use `replace` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct. +- **`lines` must not extend past `end`.** `lines` replaces exactly `[pos, end)`. The `end` line and everything after it survives. If you include `end`-line content in `lines`, it will appear twice. - `lines` entries **MUST** be literal file content with indentation copied exactly from the `read` output. If the file uses tabs, use a real tab character. - After any successful `edit` call on a file, the next change to that same file **MUST** start with a fresh `read`. Do not chain a second `edit` call off stale mental state, even if the intended range is nearby. - If you need a second change in the same local region, default to one wider `replace` over the whole owned block instead of a sequence of micro-edits on adjacent lines. Repeated small patches in a moving region are unstable. diff --git a/packages/coding-agent/src/tools/bash-interceptor.ts b/packages/coding-agent/src/tools/bash-interceptor.ts index e78096ba8..8fca9e981 100644 --- a/packages/coding-agent/src/tools/bash-interceptor.ts +++ b/packages/coding-agent/src/tools/bash-interceptor.ts @@ -5,45 +5,7 @@ * this interceptor provides helpful error messages directing them to use * the specialized tools instead. */ -import type { BashInterceptorRule } from "../config/settings-schema"; - -export const DEFAULT_BASH_INTERCEPTOR_RULES: BashInterceptorRule[] = [ - { - pattern: "^\\s*(cat|head|tail|less|more)\\s+", - tool: "read", - message: "Use the `read` tool instead of cat/head/tail. It provides better context and handles binary files.", - }, - { - pattern: "^\\s*(grep|rg|ripgrep|ag|ack)\\s+", - tool: "grep", - message: "Use the `grep` tool instead of grep/rg. It respects .gitignore and provides structured output.", - }, - { - pattern: "^\\s*(find|fd|locate)\\s+.*(-name|-iname|-type|--type|-glob)", - tool: "find", - message: "Use the `find` tool instead of find/fd. It respects .gitignore and is faster for glob patterns.", - }, - { - pattern: "^\\s*sed\\s+(-i|--in-place)", - tool: "edit", - message: "Use the `edit` tool instead of sed -i. It provides diff preview and fuzzy matching.", - }, - { - pattern: "^\\s*perl\\s+.*-[pn]?i", - tool: "edit", - message: "Use the `edit` tool instead of perl -i. It provides diff preview and fuzzy matching.", - }, - { - pattern: "^\\s*awk\\s+.*-i\\s+inplace", - tool: "edit", - message: "Use the `edit` tool instead of awk -i inplace. It provides diff preview and fuzzy matching.", - }, - { - pattern: "^\\s*(echo|printf|cat\\s*<<)\\s+.*[^|]>\\s*\\S", - tool: "write", - message: "Use the `write` tool instead of echo/cat redirection. It handles encoding and provides confirmation.", - }, -]; +import { type BashInterceptorRule, DEFAULT_BASH_INTERCEPTOR_RULES } from "../config/settings-schema"; export interface InterceptionResult { /** If true, the bash command should be blocked */ diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 0e6794de4..f78f35cd8 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -253,18 +253,29 @@ describe("applyHashlineEdits — replace", () => { expect(result.firstChangedLine).toBe(2); }); - it("range replace (shrink)", () => { + it("range replace (shrink) — end is exclusive", () => { const content = "aaa\nbbb\nccc\nddd"; + // end points to "ccc" which survives; only line 2 ("bbb") is consumed const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["ONE"] }]; + const result = applyHashlineEdits(content, edits); + expect(result.lines).toBe("aaa\nONE\nccc\nddd"); + }); + + it("range replace consuming multiple lines", () => { + const content = "aaa\nbbb\nccc\nddd"; + // end points to "ddd" which survives; lines 2-3 are consumed + const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: ["ONE"] }]; + const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nONE\nddd"); }); it("range replace (same count)", () => { const content = "aaa\nbbb\nccc\nddd"; + // Consume lines 2-3, replace with two lines. "ddd" survives. const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["XXX", "YYY"] }, + { op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: ["XXX", "YYY"] }, ]; const result = applyHashlineEdits(content, edits); @@ -272,6 +283,28 @@ describe("applyHashlineEdits — replace", () => { expect(result.firstChangedLine).toBe(2); }); + it("applies prepend at the surviving end boundary before the enclosing range replace", () => { + const content = "a\nb\nc\nd"; + const edits: HashlineEdit[] = [ + { op: "replace", pos: makeTag(2, "b"), end: makeTag(4, "d"), lines: ["X"] }, + { op: "prepend", pos: makeTag(4, "d"), lines: ["Y"] }, + ]; + const result = applyHashlineEdits(content, edits); + expect(result.lines).toBe("a\nX\nY\nd"); + expect(result.firstChangedLine).toBe(2); + }); + + it("applies single-line replace at the surviving end boundary before the enclosing range replace", () => { + const content = "a\nb\nc\nd"; + const edits: HashlineEdit[] = [ + { op: "replace", pos: makeTag(2, "b"), end: makeTag(4, "d"), lines: ["X"] }, + { op: "replace", pos: makeTag(4, "d"), lines: ["Y"] }, + ]; + const result = applyHashlineEdits(content, edits); + expect(result.lines).toBe("a\nX\nY"); + expect(result.firstChangedLine).toBe(2); + }); + it("replaces first line", () => { const content = "first\nsecond\nthird"; const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(1, "first"), lines: ["FIRST"] }]; @@ -305,9 +338,10 @@ describe("applyHashlineEdits — delete", () => { expect(result.firstChangedLine).toBe(2); }); - it("deletes range of lines", () => { + it("deletes range of lines (end exclusive)", () => { const content = "aaa\nbbb\nccc\nddd"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: [] }]; + // end points to "ddd" which survives; lines 2-3 deleted + const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nddd"); @@ -494,11 +528,12 @@ describe("applyHashlineEdits — heuristics", () => { it("does not override model whitespace choices in replacement content", () => { const content = ["import { foo } from 'x';", "import { bar } from 'y';", "const x = 1;"].join("\n"); + // end points to "const x = 1;" (survives); lines 1-2 are consumed const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "import { foo } from 'x';"), - end: makeTag(2, "import { bar } from 'y';"), + end: makeTag(3, "const x = 1;"), lines: ["import {foo} from 'x';", "import { bar } from 'y';", "// added"], }, ]; @@ -511,21 +546,21 @@ describe("applyHashlineEdits — heuristics", () => { expect(outLines[3]).toBe("const x = 1;"); }); - it("treats same-line ranges as single-line replacements", () => { + it("rejects same-line range (pos must be < end)", () => { const content = "aaa\nbbb\nccc"; const good = makeTag(2, "bbb"); const edits: HashlineEdit[] = [{ op: "replace", pos: good, end: good, lines: ["BBB"] }]; - const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("aaa\nBBB\nccc"); + expect(() => applyHashlineEdits(content, edits)).toThrow(/must be < end/); }); - it("auto-corrects off-by-one range end that duplicates a closing brace", () => { + it("auto-corrects when model re-emits the exclusive end line (closing brace)", () => { const content = "if (ok) {\n run();\n}\nafter();"; + // end points to "}" which should survive. Model accidentally includes it in lines. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "if (ok) {"), - end: makeTag(2, " run();"), + end: makeTag(3, "}"), lines: ["if (ok) {", " runSafe();", "}"], }, ]; @@ -536,13 +571,14 @@ describe("applyHashlineEdits — heuristics", () => { expect(result.warnings?.[0]).toContain('"}"'); }); - it('auto-corrects off-by-one range end that duplicates a ");" closer', () => { + it('auto-corrects when model re-emits the exclusive end line (");" closer)', () => { const content = "doThing(\n value,\n);\nnext();"; + // end points to ");" which should survive. Model accidentally includes it. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "doThing("), - end: makeTag(2, " value,"), + end: makeTag(3, ");"), lines: ["doThing(", " normalize(value),", ");"], }, ]; @@ -552,13 +588,14 @@ describe("applyHashlineEdits — heuristics", () => { expect(result.warnings?.[0]).toContain('");"'); }); - it("auto-corrects duplicated trailing lines when they match the next surviving line", () => { + it("auto-corrects when model re-emits the exclusive end line (generic content)", () => { const content = "start\n oldCall();\nnextCall();\nafter();"; + // end points to "nextCall();" which should survive. Model includes it in lines. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "start"), - end: makeTag(2, " oldCall();"), + end: makeTag(3, "nextCall();"), lines: ["start", " newCall();", "nextCall();"], }, ]; @@ -570,15 +607,18 @@ describe("applyHashlineEdits — heuristics", () => { it("auto-corrects off-by-one range start that duplicates a preceding line", () => { const content = "if (x) {\n oldBody();\n}\nafter();"; + // pos is body line 2, end is "}" (exclusive, survives). + // Model includes "if (x) {" in lines, duplicating preceding line. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(2, " oldBody();"), end: makeTag(3, "}"), - lines: ["if (x) {", " newBody();", "}"], + lines: ["if (x) {", " newBody();"], }, ]; const result = applyHashlineEdits(content, edits); + // Leading "if (x) {" is popped; "}" survives from exclusive end expect(result.lines).toBe("if (x) {\n newBody();\n}\nafter();"); expect(result.warnings).toHaveLength(1); expect(result.warnings?.[0]).toContain("removed leading replacement line"); @@ -685,11 +725,12 @@ describe("applyHashlineEdits — multiple edits", () => { it("applies non-overlapping edits against original anchors when line counts change", () => { const content = "one\ntwo\nthree\nfour\nfive\nsix"; + // end is exclusive: "four" survives, lines 2-3 are consumed const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(2, "two"), - end: makeTag(3, "three"), + end: makeTag(4, "four"), lines: ["TWO_THREE"], }, { op: "replace", pos: makeTag(6, "six"), lines: ["SIX"] }, @@ -802,7 +843,7 @@ describe("applyHashlineEdits — errors", () => { expect(() => applyHashlineEdits(content, edits)).toThrow(/does not exist/); }); - it("rejects range with start > end", () => { + it("rejects range with start >= end (exclusive end)", () => { const content = "aaa\nbbb\nccc\nddd\neee"; const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(5, "eee"), end: makeTag(2, "bbb"), lines: ["X"] }]; diff --git a/packages/coding-agent/test/modes/controllers/session-selector-delete.test.ts b/packages/coding-agent/test/modes/controllers/session-selector-delete.test.ts index fcb22edc8..cb5d8d382 100644 --- a/packages/coding-agent/test/modes/controllers/session-selector-delete.test.ts +++ b/packages/coding-agent/test/modes/controllers/session-selector-delete.test.ts @@ -1,19 +1,11 @@ -import { afterEach, describe, expect, it, mock, vi } from "bun:test"; +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import { SessionSelectorComponent } from "../../../src/modes/components/session-selector"; +import { initTheme } from "../../../src/modes/theme/theme"; import type { SessionInfo } from "../../../src/session/session-manager"; -const themeModulePath = new URL("../../../src/modes/theme/theme.ts", import.meta.url).pathname; - -mock.module(themeModulePath, () => ({ - theme: { - fg: (_tone: string, text: string) => text, - bold: (text: string) => text, - nav: { cursor: ">" }, - sep: { dot: "·" }, - boxSharp: { horizontal: "-", vertical: "|" }, - }, -})); - -import { SessionSelectorComponent } from "../../../src/modes/components/session-selector"; +beforeAll(() => { + initTheme(); +}); afterEach(() => { vi.restoreAllMocks(); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 354cbeddd..fec75e389 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { DEFAULT_BASH_INTERCEPTOR_RULES, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { EditTool } from "@oh-my-pi/pi-coding-agent/patch"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { BashTool } from "@oh-my-pi/pi-coding-agent/tools/bash"; @@ -455,6 +455,91 @@ function b() { expect(result.details).toBeUndefined(); }); + it("should expose built-in interceptor defaults truthfully", () => { + const defaultSettings = Settings.isolated({ "bashInterceptor.enabled": true }); + const explicitEmptySettings = Settings.isolated({ + "bashInterceptor.enabled": true, + "bashInterceptor.patterns": [], + }); + + expect(defaultSettings.get("bashInterceptor.patterns")).toEqual(DEFAULT_BASH_INTERCEPTOR_RULES); + expect(defaultSettings.getBashInterceptorRules()).toEqual(DEFAULT_BASH_INTERCEPTOR_RULES); + expect(explicitEmptySettings.get("bashInterceptor.patterns")).toEqual([]); + expect(explicitEmptySettings.getBashInterceptorRules()).toEqual([]); + }); + + it("should block built-in interceptor commands when enabled with default patterns", async () => { + const interceptedBashTool = wrapToolWithMetaNotice( + new BashTool(createTestToolSession(testDir, Settings.isolated({ "bashInterceptor.enabled": true }))), + ); + + await expect( + interceptedBashTool.execute( + "test-call-8-intercept-default", + { command: "cat test.txt" }, + undefined, + undefined, + { toolNames: ["read"] }, + ), + ).rejects.toThrow(/Use the `read` tool instead of cat\/head\/tail/); + }); + + it("should allow an explicit empty interceptor pattern list", async () => { + const allowedFile = path.join(testDir, "allow-empty.txt"); + fs.writeFileSync(allowedFile, "empty means empty\n"); + + const interceptedBashTool = wrapToolWithMetaNotice( + new BashTool( + createTestToolSession( + testDir, + Settings.isolated({ + "bashInterceptor.enabled": true, + "bashInterceptor.patterns": [], + }), + ), + ), + ); + + const result = await interceptedBashTool.execute( + "test-call-8-intercept-empty", + { command: `cat ${allowedFile}` }, + undefined, + undefined, + { toolNames: ["read"] }, + ); + + expect(getTextOutput(result)).toContain("empty means empty"); + }); + + it("should honor custom bash interceptor patterns", async () => { + const interceptedBashTool = wrapToolWithMetaNotice( + new BashTool( + createTestToolSession( + testDir, + Settings.isolated({ + "bashInterceptor.enabled": true, + "bashInterceptor.patterns": [ + { + pattern: "^\\s*customcmd\\s+", + tool: "grep", + message: "Use the `grep` tool for customcmd.", + }, + ], + }), + ), + ), + ); + await expect( + interceptedBashTool.execute( + "test-call-8-intercept-custom", + { command: "customcmd foo" }, + undefined, + undefined, + { toolNames: ["grep"] }, + ), + ).rejects.toThrow(/Use the `grep` tool for customcmd\./); + }); + it("should expose env values without shell re-parsing", async () => { const mermaid = [ "flowchart TD", From c11cd9023a9205ba980eb993b0453701554dcb49 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 13:56:57 +0100 Subject: [PATCH 03/22] test(tools): added createTestToolContext helper for consistent tool testing - Added createTestToolContext helper to construct AgentToolContext instances for tool tests. - Updated 2 test cases to use createTestToolContext instead of inline object literals for consistency. - Added imports for AgentToolContext and SessionManager to support new test helper. --- packages/coding-agent/test/tools.test.ts | 24 +++++++++++++++++++++--- 1 file changed, 21 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index fec75e389..b34882dd0 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -2,8 +2,10 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; +import type { AgentToolContext } from "@oh-my-pi/pi-agent-core"; import { DEFAULT_BASH_INTERCEPTOR_RULES, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { EditTool } from "@oh-my-pi/pi-coding-agent/patch"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { BashTool } from "@oh-my-pi/pi-coding-agent/tools/bash"; import { FindTool } from "@oh-my-pi/pi-coding-agent/tools/find"; @@ -42,6 +44,22 @@ function createTestToolSession(cwd: string, settings: Settings = Settings.isolat }; } +function createTestToolContext(toolNames: string[]): AgentToolContext { + return { + sessionManager: SessionManager.inMemory(), + modelRegistry: { + find: () => undefined, + getAll: () => [], + getApiKey: async () => undefined, + } as unknown as AgentToolContext["modelRegistry"], + model: undefined, + isIdle: () => true, + hasQueuedMessages: () => false, + abort: () => {}, + toolNames, + } as AgentToolContext; +} + describe("Coding Agent Tools", () => { let testDir: string; let session: ToolSession; @@ -479,7 +497,7 @@ function b() { { command: "cat test.txt" }, undefined, undefined, - { toolNames: ["read"] }, + createTestToolContext(["read"]), ), ).rejects.toThrow(/Use the `read` tool instead of cat\/head\/tail/); }); @@ -505,7 +523,7 @@ function b() { { command: `cat ${allowedFile}` }, undefined, undefined, - { toolNames: ["read"] }, + createTestToolContext(["read"]), ); expect(getTextOutput(result)).toContain("empty means empty"); @@ -535,7 +553,7 @@ function b() { { command: "customcmd foo" }, undefined, undefined, - { toolNames: ["grep"] }, + createTestToolContext(["grep"]), ), ).rejects.toThrow(/Use the `grep` tool for customcmd\./); }); From c0f57115880b99321c7bca2147a84e12ac97adfe Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 15:39:28 +0100 Subject: [PATCH 04/22] revert: exclusive end --- packages/coding-agent/CHANGELOG.md | 8 +- packages/coding-agent/src/patch/hashline.ts | 40 ++++------ .../src/prompts/tools/hashline.md | 54 ++++++++----- .../coding-agent/test/core/hashline.test.ts | 75 +++++-------------- 4 files changed, 66 insertions(+), 111 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 704a1efde..9925510b6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Added - Added ACP (Agent Client Protocol) mode for headless agent operation via `--mode acp` @@ -9,16 +10,11 @@ ### Changed -- Updated bash interceptor configuration to use customizable pattern rules instead of individual boolean flags -- Clarified hashline range replace semantics: `end` parameter is now strictly exclusive (the line it points to survives and is not consumed) - Updated ask tool rendering to support markdown formatting in questions and option labels - Refactored hook input and selector components to render titles as markdown for richer text formatting - Changed session collection to include sessions with zero messages, enabling ACP mode to create discoverable sessions immediately - Changed session persistence logic to use atomic file rewrite when flushing unflushed sessions to prevent duplication - -### Fixed - -- Fixed bash interceptor to apply built-in default rules when no custom patterns are configured +- Removed hashline edit autocorrection for duplicated boundary lines; escaped-tab autocorrection remains available for leading `\\t` sequences ## [13.14.0] - 2026-03-20 diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 493770dbc..8bd67ddfe 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -15,13 +15,6 @@ import type { HashMismatch } from "./types"; export type Anchor = { line: number; hash: string }; -/** - * Edit operation on hashline-addressed content. - * - * For range replace: `pos` is inclusive (first consumed line), - * `end` is **exclusive** (first surviving line after the range). - * The consumed range is `[pos.line, end.line - 1]`. - */ export type HashlineEdit = | { op: "replace"; pos: Anchor; end?: Anchor; lines: string[] } | { op: "append"; pos?: Anchor; lines: string[] } @@ -477,8 +470,9 @@ function shouldAutocorrect(line: string, otherLine: string): boolean { /** * Apply an array of hashline edits to file content. * - * For range replace, `end` is **exclusive**: the consumed range is - * `[pos.line, end.line - 1]` and the line at `end` survives. + * Each edit operation identifies target lines directly (`replace`, + * `append`, `prepend`). Line references are resolved via {@link parseTag} + * and hashes validated before any mutation. * * Edits are sorted bottom-up (highest effective line first) so earlier * splices don't invalidate later line numbers. @@ -524,10 +518,8 @@ export function applyHashlineEdits( const startValid = validateRef(edit.pos); const endValid = validateRef(edit.end); if (!startValid || !endValid) continue; - if (edit.pos.line >= edit.end.line) { - throw new Error( - `Range start line ${edit.pos.line} must be < end line ${edit.end.line} (end is exclusive)`, - ); + if (edit.pos.line > edit.end.line) { + throw new Error(`Range start line ${edit.pos.line} must be <= end line ${edit.end.line}`); } } else { if (!validateRef(edit.pos)) continue; @@ -605,13 +597,10 @@ export function applyHashlineEdits( case "replace": if (!edit.end) { sortLine = edit.pos.line; - precedence = 0; } else { sortLine = edit.end.line; - // Range replaces must run after edits anchored on the surviving end line, - // so those line-number references still point at the same survivor. - precedence = 3; } + precedence = 0; break; case "append": sortLine = edit.pos ? edit.pos.line : fileLines.length + 1; @@ -645,22 +634,19 @@ export function applyHashlineEdits( fileLines.splice(edit.pos.line - 1, 1, ...newLines); trackFirstChanged(edit.pos.line); } else { - // end is exclusive: consumed range is [pos.line, end.line - 1] - const count = edit.end.line - edit.pos.line; + const count = edit.end.line - edit.pos.line + 1; const newLines = [...edit.lines]; - // The end line itself survives (exclusive). If the model re-emits it - // in lines, that's a duplication mistake — auto-correct by popping. const trailingReplacementLine = newLines[newLines.length - 1]?.trimEnd(); - const nextSurvivingLine = fileLines[edit.end.line - 1]?.trimEnd(); + const nextSurvivingLine = fileLines[edit.end.line]?.trimEnd(); if ( shouldAutocorrect(trailingReplacementLine, nextSurvivingLine) && - // Safety: only correct when the last consumed line differs from the duplicate. - // If the last consumed line is the same as the surviving end line, it's coincidence. - fileLines[edit.end.line - 2]?.trimEnd() !== trailingReplacementLine + // Safety: only correct when end-line content differs from the duplicate. + // If end already points to the boundary, matching next line is coincidence. + fileLines[edit.end.line - 1]?.trimEnd() !== trailingReplacementLine ) { newLines.pop(); warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}..${edit.end.line}#${edit.end.hash}: removed trailing replacement line "${trailingReplacementLine}" that duplicated the surviving end line`, + `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed trailing replacement line "${trailingReplacementLine}" that duplicated next surviving line`, ); } const leadingReplacementLine = newLines[0]?.trimEnd(); @@ -673,7 +659,7 @@ export function applyHashlineEdits( ) { newLines.shift(); warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}..${edit.end.line}#${edit.end.hash}: removed leading replacement line "${leadingReplacementLine}" that duplicated preceding surviving line`, + `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed leading replacement line "${leadingReplacementLine}" that duplicated preceding surviving line`, ); } fileLines.splice(edit.pos.line - 1, count, ...newLines); diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index 821564d68..2b7fea2e3 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -11,13 +11,14 @@ Read the file first to get fresh tags. Submit one `edit` call per file with all - if `replace`: first line to rewrite - if `prepend`: line to insert new lines **before**; omit for beginning of file - if `append`: line to insert new lines **after**; omit for end of file -**`edits[n].end`** — range replace only. The first line **after** the range (exclusive — this line survives). Omit for single-line replace. +**`edits[n].end`** — range replace only. The last line of the range (inclusive). Omit for single-line replace. **`edits[n].lines`** — the replacement content: - - for `replace`: the lines that will replace `[pos, end)`. Everything from `pos` up to (but not including) `end` is removed; `lines` is inserted in its place. + - for `replace`: the exact lines that will replace `[pos, end??pos]` inclusively (or the single `pos` line when `end` is omitted) - for `prepend`/`append`: the new lines to insert - `[""]` — blank line - `null` or `[]` — delete if replace -- **`end` is exclusive — the line it points to stays in the file.** You do not need to re-emit it in `lines`. If you accidentally include it in `lines`, it will be duplicated. +- If `lines` contains content that already exists after `end`, those lines **will be duplicated** in the output. +- Keep `lines` to exactly what belongs inside the consumed range. - Ops are applied bottom-up. Tags **MUST** be referenced from the most recent `read` output. @@ -70,22 +71,22 @@ Single line — `lines: null` deletes entirely: }] } ``` -Range — remove the legacy block (lines 10–11). `end` points to line 12 (the line after the range): +Range — remove the legacy block (lines 10–11): ``` { path: "util.ts", edits: [{ op: "replace", pos: {{hlineref 10 "\t// TODO: remove after migration"}}, - end: {{hlineref 12 "\ttry {"}}, + end: {{hlineref 11 "\tlegacy();"}}, lines: null }] } ``` - -Replace the catch body with smarter error handling. `pos` is the first body line, `end` is the closer — the closer survives automatically. + +Replace the catch body with smarter error handling. Shape (a): `pos` is the first body line, `end` is the last body line. The catch header (line 14) and its closer (line 17) are outside the range and stay untouched. When changing body content, replace the **entire** body span — not just one line inside it. Patching one line leaves the rest of the body stale. ``` @@ -94,7 +95,7 @@ When changing body content, replace the **entire** body span — not just one li edits: [{ op: "replace", pos: {{hlineref 15 "\t\tconsole.error(err);"}}, - end: {{hlineref 17 "\t}"}}, + end: {{hlineref 16 "\t\treturn null;"}}, lines: [ "\t\tif (isEnoent(err)) return null;", "\t\tthrow err;" @@ -102,15 +103,28 @@ When changing body content, replace the **entire** body span — not just one li }] } ``` -Result: lines 15–16 are replaced. The `\t}` on line 17 stays because `end` is exclusive. - -Simplify `beta()` to a one-liner. `pos`=header (consumed), `end`=the line **after** the block (survives). + +Simplify `beta()` to a one-liner. Shape (b): `pos`=header, `end`=closer, re-emit all in `lines`. -Since `}` on line 18 is the last line of `beta()` and we want to consume it, `end` must point to the next line after the block. When line 18 is the last line of the file, omit `end` — single-line replace plus a delete of lines 10–17 first, or use `write` to rewrite the file. - -When there IS a line after the block: +Bad — `end` stops at the inner `\t}` on line 17, so the outer `}` on line 18 survives. Result: two consecutive `}` lines. +``` +{ + path: "util.ts", + edits: [{ + op: "replace", + pos: {{hlineref 9 "function beta() {"}}, + end: {{hlineref 17 "\t}"}}, + lines: [ + "function beta() {", + "\treturn parse(data);", + "}" + ] + }] +} +``` +Good — `end` includes the function's own `}` on line 18, so the old closer is consumed: ``` { path: "util.ts", @@ -120,12 +134,12 @@ When there IS a line after the block: end: {{hlineref 18 "}"}}, lines: [ "function beta() {", - "\treturn parse(data);" + "\treturn parse(data);", + "}" ] }] } ``` -Result: lines 9–17 are consumed (replaced). `}` on line 18 survives as beta's closer — no need to re-emit it. @@ -135,7 +149,7 @@ Bad — if you need to change code on both sides of that line, replacing just th Good — choose one of two safe shapes instead: - move inward and replace only body-owned lines -- expand outward and replace one whole owned block +- expand outward and replace one whole owned block, consuming its real closer/separator too @@ -166,9 +180,9 @@ Use a trailing `""` to preserve the blank line between sibling declarations. - For `append`/`prepend`, `lines` **MUST** contain only the newly introduced content. Do not re-emit surrounding content, or terminators that already exist. - When changing existing code near a block tail or closing delimiter, default to `replace` over the owned span instead of inserting around the boundary. - When adding a sibling declaration, default to `prepend` on the next sibling declaration instead of `append` on the previous block's closing brace. -- **`end` is the boundary you want to keep.** Point `end` at the closing delimiter (`}`, `)`, ``) when you want it to survive. Point `end` past it when you want to consume it. Do **not** include the `end` line in `lines` — it survives on its own. -- **Never target shared boundary lines.** Do not use `replace` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct. -- **`lines` must not extend past `end`.** `lines` replaces exactly `[pos, end)`. The `end` line and everything after it survives. If you include `end`-line content in `lines`, it will appear twice. +- **Block boundaries travel together.** For a block `{ header / body / closer }`, there are exactly two valid replace shapes: (a) replace only the body — `pos`=first body line, `end`=last body line, leave the header and closer untouched; or (b) replace the whole block — `pos`=header, `end`=closer, re-emit all three in `lines`. Never split them: do not set `end` to the closer while omitting it from `lines` (deletes it), and do not emit the closer in `lines` without including it in `end` (duplicates it). This applies to every block terminator: `}`, `continue`, `break`, `return`, `throw`. +- **Never target shared boundary lines.** Do not use `replace` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct including its true trailing delimiter. +- **`lines` must not extend past `end`.** `lines` replaces exactly `pos..end`. Content after `end` survives. If you include lines in `lines` that exist after `end`, they will appear twice. Either extend `end` to cover all lines you are re-emitting, or remove the extra lines from `lines`. - `lines` entries **MUST** be literal file content with indentation copied exactly from the `read` output. If the file uses tabs, use a real tab character. - After any successful `edit` call on a file, the next change to that same file **MUST** start with a fresh `read`. Do not chain a second `edit` call off stale mental state, even if the intended range is nearby. - If you need a second change in the same local region, default to one wider `replace` over the whole owned block instead of a sequence of micro-edits on adjacent lines. Repeated small patches in a moving region are unstable. diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index f78f35cd8..0e6794de4 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -253,29 +253,18 @@ describe("applyHashlineEdits — replace", () => { expect(result.firstChangedLine).toBe(2); }); - it("range replace (shrink) — end is exclusive", () => { + it("range replace (shrink)", () => { const content = "aaa\nbbb\nccc\nddd"; - // end points to "ccc" which survives; only line 2 ("bbb") is consumed const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["ONE"] }]; - const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("aaa\nONE\nccc\nddd"); - }); - - it("range replace consuming multiple lines", () => { - const content = "aaa\nbbb\nccc\nddd"; - // end points to "ddd" which survives; lines 2-3 are consumed - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: ["ONE"] }]; - const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nONE\nddd"); }); it("range replace (same count)", () => { const content = "aaa\nbbb\nccc\nddd"; - // Consume lines 2-3, replace with two lines. "ddd" survives. const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: ["XXX", "YYY"] }, + { op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["XXX", "YYY"] }, ]; const result = applyHashlineEdits(content, edits); @@ -283,28 +272,6 @@ describe("applyHashlineEdits — replace", () => { expect(result.firstChangedLine).toBe(2); }); - it("applies prepend at the surviving end boundary before the enclosing range replace", () => { - const content = "a\nb\nc\nd"; - const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "b"), end: makeTag(4, "d"), lines: ["X"] }, - { op: "prepend", pos: makeTag(4, "d"), lines: ["Y"] }, - ]; - const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("a\nX\nY\nd"); - expect(result.firstChangedLine).toBe(2); - }); - - it("applies single-line replace at the surviving end boundary before the enclosing range replace", () => { - const content = "a\nb\nc\nd"; - const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "b"), end: makeTag(4, "d"), lines: ["X"] }, - { op: "replace", pos: makeTag(4, "d"), lines: ["Y"] }, - ]; - const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("a\nX\nY"); - expect(result.firstChangedLine).toBe(2); - }); - it("replaces first line", () => { const content = "first\nsecond\nthird"; const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(1, "first"), lines: ["FIRST"] }]; @@ -338,10 +305,9 @@ describe("applyHashlineEdits — delete", () => { expect(result.firstChangedLine).toBe(2); }); - it("deletes range of lines (end exclusive)", () => { + it("deletes range of lines", () => { const content = "aaa\nbbb\nccc\nddd"; - // end points to "ddd" which survives; lines 2-3 deleted - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(4, "ddd"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nddd"); @@ -528,12 +494,11 @@ describe("applyHashlineEdits — heuristics", () => { it("does not override model whitespace choices in replacement content", () => { const content = ["import { foo } from 'x';", "import { bar } from 'y';", "const x = 1;"].join("\n"); - // end points to "const x = 1;" (survives); lines 1-2 are consumed const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "import { foo } from 'x';"), - end: makeTag(3, "const x = 1;"), + end: makeTag(2, "import { bar } from 'y';"), lines: ["import {foo} from 'x';", "import { bar } from 'y';", "// added"], }, ]; @@ -546,21 +511,21 @@ describe("applyHashlineEdits — heuristics", () => { expect(outLines[3]).toBe("const x = 1;"); }); - it("rejects same-line range (pos must be < end)", () => { + it("treats same-line ranges as single-line replacements", () => { const content = "aaa\nbbb\nccc"; const good = makeTag(2, "bbb"); const edits: HashlineEdit[] = [{ op: "replace", pos: good, end: good, lines: ["BBB"] }]; - expect(() => applyHashlineEdits(content, edits)).toThrow(/must be < end/); + const result = applyHashlineEdits(content, edits); + expect(result.lines).toBe("aaa\nBBB\nccc"); }); - it("auto-corrects when model re-emits the exclusive end line (closing brace)", () => { + it("auto-corrects off-by-one range end that duplicates a closing brace", () => { const content = "if (ok) {\n run();\n}\nafter();"; - // end points to "}" which should survive. Model accidentally includes it in lines. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "if (ok) {"), - end: makeTag(3, "}"), + end: makeTag(2, " run();"), lines: ["if (ok) {", " runSafe();", "}"], }, ]; @@ -571,14 +536,13 @@ describe("applyHashlineEdits — heuristics", () => { expect(result.warnings?.[0]).toContain('"}"'); }); - it('auto-corrects when model re-emits the exclusive end line (");" closer)', () => { + it('auto-corrects off-by-one range end that duplicates a ");" closer', () => { const content = "doThing(\n value,\n);\nnext();"; - // end points to ");" which should survive. Model accidentally includes it. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "doThing("), - end: makeTag(3, ");"), + end: makeTag(2, " value,"), lines: ["doThing(", " normalize(value),", ");"], }, ]; @@ -588,14 +552,13 @@ describe("applyHashlineEdits — heuristics", () => { expect(result.warnings?.[0]).toContain('");"'); }); - it("auto-corrects when model re-emits the exclusive end line (generic content)", () => { + it("auto-corrects duplicated trailing lines when they match the next surviving line", () => { const content = "start\n oldCall();\nnextCall();\nafter();"; - // end points to "nextCall();" which should survive. Model includes it in lines. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(1, "start"), - end: makeTag(3, "nextCall();"), + end: makeTag(2, " oldCall();"), lines: ["start", " newCall();", "nextCall();"], }, ]; @@ -607,18 +570,15 @@ describe("applyHashlineEdits — heuristics", () => { it("auto-corrects off-by-one range start that duplicates a preceding line", () => { const content = "if (x) {\n oldBody();\n}\nafter();"; - // pos is body line 2, end is "}" (exclusive, survives). - // Model includes "if (x) {" in lines, duplicating preceding line. const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(2, " oldBody();"), end: makeTag(3, "}"), - lines: ["if (x) {", " newBody();"], + lines: ["if (x) {", " newBody();", "}"], }, ]; const result = applyHashlineEdits(content, edits); - // Leading "if (x) {" is popped; "}" survives from exclusive end expect(result.lines).toBe("if (x) {\n newBody();\n}\nafter();"); expect(result.warnings).toHaveLength(1); expect(result.warnings?.[0]).toContain("removed leading replacement line"); @@ -725,12 +685,11 @@ describe("applyHashlineEdits — multiple edits", () => { it("applies non-overlapping edits against original anchors when line counts change", () => { const content = "one\ntwo\nthree\nfour\nfive\nsix"; - // end is exclusive: "four" survives, lines 2-3 are consumed const edits: HashlineEdit[] = [ { op: "replace", pos: makeTag(2, "two"), - end: makeTag(4, "four"), + end: makeTag(3, "three"), lines: ["TWO_THREE"], }, { op: "replace", pos: makeTag(6, "six"), lines: ["SIX"] }, @@ -843,7 +802,7 @@ describe("applyHashlineEdits — errors", () => { expect(() => applyHashlineEdits(content, edits)).toThrow(/does not exist/); }); - it("rejects range with start >= end (exclusive end)", () => { + it("rejects range with start > end", () => { const content = "aaa\nbbb\nccc\nddd\neee"; const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(5, "eee"), end: makeTag(2, "bbb"), lines: ["X"] }]; From 2771c2399d3899370611c50b41691f78252559a5 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 16:50:13 +0100 Subject: [PATCH 05/22] feat(autoresearch): introduced autonomous experiment loop with metric-driven optimization - Added autoresearch extension with autonomous experiment loop supporting init, run, and log experiment tools for metric-driven optimization. - Added widget placement system enabling extensions to position UI components above or below the editor via ExtensionWidgetOptions. - Added dashboard controller with interactive overlay for viewing experiment results, metrics, and progress with keyboard navigation. - Removed auto-correction logic for off-by-one range edits in hashline editor to preserve user intent in patch operations. - Added state reconstruction utilities to parse autoresearch.jsonl logs and rebuild experiment state across sessions. - Added comprehensive type definitions and helper utilities for metric parsing, ASI validation, and process management. --- packages/coding-agent/CHANGELOG.md | 28 + .../src/autoresearch/command-start.md | 10 + .../src/autoresearch/dashboard.ts | 341 +++++++++++++ .../coding-agent/src/autoresearch/helpers.ts | 202 ++++++++ .../coding-agent/src/autoresearch/index.ts | 205 ++++++++ .../coding-agent/src/autoresearch/prompt.md | 155 ++++++ .../src/autoresearch/resume-message.md | 10 + .../coding-agent/src/autoresearch/state.ts | 279 ++++++++++ .../src/autoresearch/tools/init-experiment.ts | 117 +++++ .../src/autoresearch/tools/log-experiment.ts | 419 +++++++++++++++ .../src/autoresearch/tools/run-experiment.ts | 478 ++++++++++++++++++ .../coding-agent/src/autoresearch/types.ts | 167 ++++++ .../src/extensibility/extensions/types.ts | 21 +- .../controllers/extension-ui-controller.ts | 93 +++- .../src/modes/interactive-mode.ts | 20 +- .../coding-agent/src/modes/rpc/rpc-mode.ts | 9 +- .../coding-agent/src/modes/rpc/rpc-types.ts | 1 + packages/coding-agent/src/modes/types.ts | 11 +- packages/coding-agent/src/patch/hashline.ts | 41 +- packages/coding-agent/src/sdk.ts | 2 + .../test/autoresearch-state.test.ts | 107 ++++ .../coding-agent/test/core/hashline.test.ts | 42 +- 22 files changed, 2670 insertions(+), 88 deletions(-) create mode 100644 packages/coding-agent/src/autoresearch/command-start.md create mode 100644 packages/coding-agent/src/autoresearch/dashboard.ts create mode 100644 packages/coding-agent/src/autoresearch/helpers.ts create mode 100644 packages/coding-agent/src/autoresearch/index.ts create mode 100644 packages/coding-agent/src/autoresearch/prompt.md create mode 100644 packages/coding-agent/src/autoresearch/resume-message.md create mode 100644 packages/coding-agent/src/autoresearch/state.ts create mode 100644 packages/coding-agent/src/autoresearch/tools/init-experiment.ts create mode 100644 packages/coding-agent/src/autoresearch/tools/log-experiment.ts create mode 100644 packages/coding-agent/src/autoresearch/tools/run-experiment.ts create mode 100644 packages/coding-agent/src/autoresearch/types.ts create mode 100644 packages/coding-agent/test/autoresearch-state.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9925510b6..fde575753 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,18 +4,46 @@ ### Added +- Added autoresearch extension with autonomous experiment loop capabilities +- Added `init_experiment` tool to initialize and reset autoresearch sessions with configurable metrics +- Added `log_experiment` tool to record experiment results with metric parsing and confidence tracking +- Added `run_experiment` tool to execute commands and capture metrics with timeout and crash detection +- Added autoresearch dashboard controller for displaying experiment results and optimization progress +- Added support for secondary metrics tracking alongside primary metric +- Added `ExtensionWidgetContent` and `ExtensionUiComponentFactory` types for flexible widget configuration +- Added `ExtensionWidgetOptions` interface with `placement` parameter to position widgets above or below editor +- Added `WidgetPlacement` type supporting 'aboveEditor' and 'belowEditor' placement options +- Added `hookWidgetContainerAbove` and `hookWidgetContainerBelow` containers to InteractiveMode for separate widget management +- Added autoresearch mode for autonomous experiment loops with init_experiment, log_experiment, and run_experiment tools +- Added autoresearch dashboard widget displaying experiment results, metrics, and optimization progress +- Added support for metric tracking with configurable direction (lower/higher is better) and secondary metrics +- Added widget placement options to position extensions above or below the editor via `placement` parameter +- Added `ExtensionWidgetContent` and `ExtensionWidgetOptions` types for flexible widget configuration - Added ACP (Agent Client Protocol) mode for headless agent operation via `--mode acp` - Added support for Agent Client Protocol SDK integration with session management, MCP server configuration, and streaming communication - Added `ensureOnDisk()` method to SessionManager to persist sessions immediately for ACP discovery ### Changed +- Changed `setWidget` API to accept `ExtensionWidgetOptions` parameter for placement control +- Changed widget placement logic to manage widgets above and below editor separately +- Changed hashline edit application to preserve duplicated boundary lines exactly as provided instead of auto-correcting them +- Updated RPC mode to support widget placement option in `setWidget` requests +- Changed hashline edit application to preserve duplicated boundary lines exactly as provided instead of auto-correcting them +- Changed widget API to support placement options and component factories in addition to string arrays +- Updated extension UI controller to manage widgets above and below the editor separately - Updated ask tool rendering to support markdown formatting in questions and option labels - Refactored hook input and selector components to render titles as markdown for richer text formatting - Changed session collection to include sessions with zero messages, enabling ACP mode to create discoverable sessions immediately - Changed session persistence logic to use atomic file rewrite when flushing unflushed sessions to prevent duplication - Removed hashline edit autocorrection for duplicated boundary lines; escaped-tab autocorrection remains available for leading `\\t` sequences +### Removed + +- Removed auto-correction of off-by-one range edits that duplicated closing braces or boundary lines +- Removed `shouldAutocorrect` function and related boundary line deduplication logic from hashline editor +- Removed auto-correction of off-by-one range edits that duplicated closing braces or boundary lines + ## [13.14.0] - 2026-03-20 ### Added diff --git a/packages/coding-agent/src/autoresearch/command-start.md b/packages/coding-agent/src/autoresearch/command-start.md new file mode 100644 index 000000000..1594a6a2c --- /dev/null +++ b/packages/coding-agent/src/autoresearch/command-start.md @@ -0,0 +1,10 @@ +Autoresearch mode is active. + +Goal: +{{goal}} + +Start or resume the autoresearch loop now. + +- Read `autoresearch.md` if it already exists. +- Otherwise create the autoresearch workspace, initialize the experiment, run a baseline, and keep iterating. +- Continue until interrupted or until the configured iteration cap is reached. diff --git a/packages/coding-agent/src/autoresearch/dashboard.ts b/packages/coding-agent/src/autoresearch/dashboard.ts new file mode 100644 index 000000000..246277baf --- /dev/null +++ b/packages/coding-agent/src/autoresearch/dashboard.ts @@ -0,0 +1,341 @@ +import { matchesKey, Text, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui"; +import type { Theme } from "../modes/theme/theme"; +import { formatElapsed, formatNum, isBetter } from "./helpers"; +import { currentResults, findBaselineMetric, findBaselineRunNumber, findBaselineSecondary } from "./state"; +import type { AutoresearchRuntime, DashboardController, ExperimentResult, ExperimentState } from "./types"; + +export function createDashboardController(): DashboardController { + let overlayTui: { requestRender(): void } | null = null; + let spinnerTimer: NodeJS.Timeout | undefined; + let spinnerFrame = 0; + + const requestRender = (): void => { + overlayTui?.requestRender(); + }; + + const clear = (): void => { + overlayTui = null; + if (spinnerTimer) { + clearInterval(spinnerTimer); + spinnerTimer = undefined; + } + }; + + return { + clear(ctx): void { + clear(); + if (ctx.hasUI) { + ctx.ui.setWidget("autoresearch", undefined); + } + }, + requestRender, + updateWidget(ctx, runtime): void { + if (!ctx.hasUI) return; + const state = runtime.state; + if (state.results.length === 0 && !runtime.runningExperiment) { + ctx.ui.setWidget("autoresearch", undefined); + return; + } + + ctx.ui.setWidget("autoresearch", (_tui, theme) => { + if (state.results.length === 0 && runtime.runningExperiment) { + return new Text(renderRunningOnly(runtime, state, theme), 0, 0); + } + if (runtime.dashboardExpanded) { + const width = process.stdout.columns ?? 120; + const lines = [ + renderExpandedHeader(state, width, theme), + ...renderDashboardLines(state, width, theme, 8), + ]; + return new Text(lines.join("\n"), 0, 0); + } + return new Text(renderCollapsedLine(runtime, state, theme), 0, 0); + }); + }, + async showOverlay(ctx, runtime): Promise { + if (!ctx.hasUI || runtime.state.results.length === 0) return; + await ctx.ui.custom( + (tui, theme, _keybindings, done) => { + overlayTui = tui; + if (!spinnerTimer) { + spinnerTimer = setInterval(() => { + spinnerFrame += 1; + requestRender(); + }, 80); + } + + let scrollOffset = 0; + return { + render(width: number): string[] { + const terminalRows = process.stdout.rows ?? 40; + const header = renderExpandedHeader(runtime.state, width, theme); + const body = renderDashboardLines(runtime.state, width, theme, 0); + if (runtime.runningExperiment) { + body.push(renderOverlayRunningLine(runtime, theme, width, spinnerFrame)); + } + const viewportRows = Math.max(4, terminalRows - 4); + const maxScroll = Math.max(0, body.length - viewportRows); + if (scrollOffset > maxScroll) scrollOffset = maxScroll; + const visible = body.slice(scrollOffset, scrollOffset + viewportRows); + const footer = renderOverlayFooter(width, scrollOffset, viewportRows, body.length, theme); + return [ + header, + ...visible, + ...Array.from({ length: Math.max(0, viewportRows - visible.length) }, () => ""), + footer, + ]; + }, + handleInput(data: string): void { + const totalRows = + renderDashboardLines(runtime.state, process.stdout.columns ?? 120, theme, 0).length + + (runtime.runningExperiment ? 1 : 0); + const viewportRows = Math.max(4, (process.stdout.rows ?? 40) - 4); + const maxScroll = Math.max(0, totalRows - viewportRows); + if (matchesKey(data, "escape") || matchesKey(data, "esc") || data === "q") { + done(undefined); + return; + } + if (matchesKey(data, "up") || data === "k") { + scrollOffset = Math.max(0, scrollOffset - 1); + } else if (matchesKey(data, "down") || data === "j") { + scrollOffset = Math.min(maxScroll, scrollOffset + 1); + } else if (matchesKey(data, "pageUp")) { + scrollOffset = Math.max(0, scrollOffset - viewportRows); + } else if (matchesKey(data, "pageDown")) { + scrollOffset = Math.min(maxScroll, scrollOffset + viewportRows); + } else if (data === "g") { + scrollOffset = 0; + } else if (data === "G") { + scrollOffset = maxScroll; + } + tui.requestRender(); + }, + invalidate(): void {}, + dispose(): void { + clear(); + }, + }; + }, + { overlay: true }, + ); + }, + }; +} + +function renderRunningOnly(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { + const parts = [theme.fg("accent", "autoresearch"), theme.fg("warning", " running...")]; + if (state.name) { + parts.push(theme.fg("dim", ` | ${state.name}`)); + } + if (runtime.runningExperiment) { + parts.push(theme.fg("dim", ` | ${runtime.runningExperiment.command}`)); + } + return parts.join(""); +} + +function renderExpandedHeader(state: ExperimentState, width: number, theme: Theme): string { + const label = state.name ? ` autoresearch: ${state.name} ` : " autoresearch "; + const hint = theme.fg("dim", " ctrl+x collapse ctrl+shift+x fullscreen "); + const fillWidth = Math.max(0, width - visibleWidth(label) - visibleWidth(hint)); + return truncateToWidth(theme.fg("accent", label) + theme.fg("borderMuted", "-".repeat(fillWidth)) + hint, width); +} + +function renderCollapsedLine(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { + const current = currentResults(state.results, state.currentSegment); + const kept = current.filter(result => result.status === "keep").length; + const crashed = current.filter(result => result.status === "crash").length; + const checksFailed = current.filter(result => result.status === "checks_failed").length; + const best = findBestResult(state); + const parts = [ + theme.fg("accent", "autoresearch"), + theme.fg("muted", ` ${state.results.length} runs`), + theme.fg("success", ` ${kept} kept`), + ]; + if (crashed > 0) parts.push(theme.fg("error", ` ${crashed} crash`)); + if (checksFailed > 0) parts.push(theme.fg("error", ` ${checksFailed} checks_failed`)); + parts.push(theme.fg("dim", " | ")); + parts.push( + theme.fg( + "warning", + `${state.metricName}: ${formatNum(best?.result.metric ?? state.bestMetric, state.metricUnit)}`, + ), + ); + if (state.confidence !== null) { + const confidenceColor = state.confidence >= 2 ? "success" : state.confidence >= 1 ? "warning" : "error"; + parts.push(theme.fg("dim", " | ")); + parts.push(theme.fg(confidenceColor, `conf ${state.confidence.toFixed(1)}x`)); + } + if (runtime.runningExperiment) { + parts.push(theme.fg("dim", ` | running ${formatElapsed(Date.now() - runtime.runningExperiment.startedAt)}`)); + } + parts.push(theme.fg("dim", " | ctrl+x expand")); + return parts.join(""); +} + +export function renderDashboardLines(state: ExperimentState, width: number, theme: Theme, maxRows: number): string[] { + if (state.results.length === 0) { + return [theme.fg("dim", "No experiments logged yet.")]; + } + + const current = currentResults(state.results, state.currentSegment); + const kept = current.filter(result => result.status === "keep").length; + const discarded = current.filter(result => result.status === "discard").length; + const crashed = current.filter(result => result.status === "crash").length; + const checksFailed = current.filter(result => result.status === "checks_failed").length; + const baseline = findBaselineMetric(state.results, state.currentSegment); + const baselineRunNumber = findBaselineRunNumber(state.results, state.currentSegment); + const baselineSecondary = findBaselineSecondary(state.results, state.currentSegment, state.secondaryMetrics); + const best = findBestResult(state); + const lines = [ + truncateToWidth( + `Runs: ${state.results.length} ${kept} kept ${discarded} discarded ${crashed} crashed ${checksFailed} checks_failed`, + width, + ), + truncateToWidth( + `Baseline: ${formatNum(baseline, state.metricUnit)}${baselineRunNumber ? ` (#${baselineRunNumber})` : ""}`, + width, + ), + ]; + if (best) { + let progress = `Best: ${formatNum(best.result.metric, state.metricUnit)} (#${best.index + 1})`; + if (baseline !== null && baseline !== 0 && best.result.metric !== baseline) { + const delta = ((best.result.metric - baseline) / baseline) * 100; + const sign = delta > 0 ? "+" : ""; + progress += ` ${sign}${delta.toFixed(1)}%`; + } + if (state.confidence !== null) { + progress += ` conf ${state.confidence.toFixed(1)}x`; + } + lines.push(truncateToWidth(progress, width)); + if (state.secondaryMetrics.length > 0) { + const details = state.secondaryMetrics + .map(metric => + renderSecondarySummary( + metric.name, + best.result.metrics[metric.name], + baselineSecondary[metric.name], + metric.unit, + ), + ) + .filter((value): value is string => Boolean(value)); + if (details.length > 0) { + lines.push(truncateToWidth(`Secondary: ${details.join(" ")}`, width)); + } + } + } + lines.push(""); + lines.push(renderTableHeader(state, width, theme)); + lines.push(theme.fg("borderMuted", "-".repeat(Math.max(0, width - 1)))); + + const visible = maxRows > 0 ? state.results.slice(-maxRows) : state.results; + if (visible.length < state.results.length) { + lines.push(theme.fg("dim", `... ${state.results.length - visible.length} earlier runs hidden ...`)); + } + for (const result of visible) { + lines.push(renderResultRow(result, state, baselineSecondary, width, theme)); + } + return lines; +} + +function renderTableHeader(state: ExperimentState, width: number, theme: Theme): string { + const secondaryHeader = state.secondaryMetrics.map(metric => truncateToWidth(metric.name, 10)).join(" "); + return truncateToWidth( + `${theme.fg("muted", "#".padEnd(4))}${theme.fg("muted", "commit".padEnd(10))}${theme.fg("warning", state.metricName.padEnd(12))}${secondaryHeader ? `${theme.fg("muted", secondaryHeader)} ` : ""}${theme.fg("muted", "status".padEnd(14))}${theme.fg("muted", "description")}`, + width, + ); +} + +function renderResultRow( + result: ExperimentResult, + state: ExperimentState, + baselineSecondary: { [key: string]: number }, + width: number, + theme: Theme, +): string { + const runNumber = state.results.indexOf(result) + 1; + const secondary = state.secondaryMetrics + .map(metric => + truncateToWidth( + renderSecondaryCell(result.metrics[metric.name], metric.unit, baselineSecondary[metric.name]), + 10, + ).padEnd(11), + ) + .join(""); + const statusColor = result.status === "keep" ? "success" : result.status === "discard" ? "warning" : "error"; + const line = + `${theme.fg("dim", String(runNumber).padEnd(4))}` + + `${theme.fg("accent", (result.commit || "-").padEnd(10))}` + + `${theme.fg(statusColor, formatNum(result.metric, state.metricUnit).padEnd(12))}` + + `${secondary}` + + `${theme.fg(statusColor, result.status.padEnd(14))}` + + `${theme.fg("muted", result.description)}`; + return truncateToWidth(line, width); +} + +function renderSecondaryCell(value: number | undefined, unit: string, baseline: number | undefined): string { + if (value === undefined) return "-"; + const formatted = formatNum(value, unit); + if (baseline === undefined || baseline === 0 || baseline === value) return formatted; + const delta = ((value - baseline) / baseline) * 100; + const sign = delta > 0 ? "+" : ""; + return `${formatted} ${sign}${delta.toFixed(1)}%`; +} + +function renderSecondarySummary( + name: string, + value: number | undefined, + baseline: number | undefined, + unit: string, +): string | null { + if (value === undefined) return null; + if (baseline === undefined || baseline === 0 || baseline === value) { + return `${name} ${formatNum(value, unit)}`; + } + const delta = ((value - baseline) / baseline) * 100; + const sign = delta > 0 ? "+" : ""; + return `${name} ${formatNum(value, unit)} ${sign}${delta.toFixed(1)}%`; +} + +function renderOverlayRunningLine( + runtime: AutoresearchRuntime, + theme: Theme, + width: number, + spinnerFrame: number, +): string { + const spinner = theme.spinnerFrames[spinnerFrame % theme.spinnerFrames.length] ?? "*"; + return truncateToWidth( + theme.fg( + "warning", + `${spinner} running ${formatElapsed(Date.now() - (runtime.runningExperiment?.startedAt ?? Date.now()))} ${runtime.runningExperiment?.command ?? ""}`, + ), + width, + ); +} + +function renderOverlayFooter( + width: number, + scrollOffset: number, + viewportRows: number, + totalRows: number, + theme: Theme, +): string { + const position = + totalRows > viewportRows + ? ` ${scrollOffset + 1}-${Math.min(totalRows, scrollOffset + viewportRows)}/${totalRows}` + : ""; + const hint = theme.fg("dim", ` up/down j/k pageup pagedown g G esc${position} `); + const fill = Math.max(0, width - visibleWidth(hint)); + return theme.fg("borderMuted", "-".repeat(fill)) + hint; +} + +function findBestResult(state: ExperimentState): { index: number; result: ExperimentResult } | null { + let best: { index: number; result: ExperimentResult } | null = null; + for (let index = 0; index < state.results.length; index += 1) { + const result = state.results[index]; + if (result.segment !== state.currentSegment || result.status !== "keep" || result.metric <= 0) continue; + if (!best || isBetter(result.metric, best.result.metric, state.bestDirection)) { + best = { index, result }; + } + } + return best; +} diff --git a/packages/coding-agent/src/autoresearch/helpers.ts b/packages/coding-agent/src/autoresearch/helpers.ts new file mode 100644 index 000000000..d5b845ef7 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/helpers.ts @@ -0,0 +1,202 @@ +import * as crypto from "node:crypto"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { isEnoent } from "@oh-my-pi/pi-utils"; +import type { ASIData, ASIValue, AutoresearchConfig, MetricDirection } from "./types"; + +export const METRIC_LINE_PREFIX = "METRIC"; +export const ASI_LINE_PREFIX = "ASI"; +export const EXPERIMENT_MAX_LINES = 10; +export const EXPERIMENT_MAX_BYTES = 4 * 1024; + +const DENIED_KEY_NAMES = new Set(["__proto__", "constructor", "prototype"]); + +export function parseMetricLines(output: string): Map { + const metrics = new Map(); + const regex = new RegExp(`^${METRIC_LINE_PREFIX}\\s+([\\w.µ-]+)=(\\S+)\\s*$`, "gm"); + let match = regex.exec(output); + while (match !== null) { + const name = match[1]; + if (!DENIED_KEY_NAMES.has(name)) { + const value = Number(match[2]); + if (Number.isFinite(value)) { + metrics.set(name, value); + } + } + match = regex.exec(output); + } + return metrics; +} + +export function parseAsiLines(output: string): ASIData | null { + const asi: ASIData = {}; + const regex = new RegExp(`^${ASI_LINE_PREFIX}\\s+([\\w.-]+)=(.+)\\s*$`, "gm"); + let match = regex.exec(output); + while (match !== null) { + const key = match[1]; + if (!DENIED_KEY_NAMES.has(key)) { + asi[key] = parseAsiValue(match[2]); + } + match = regex.exec(output); + } + return Object.keys(asi).length > 0 ? asi : null; +} + +function parseAsiValue(raw: string): ASIValue { + const value = raw.trim(); + if (value === "true") return true; + if (value === "false") return false; + if (value === "null") return null; + if (/^-?\d+(?:\.\d+)?$/.test(value)) { + const numberValue = Number(value); + if (Number.isFinite(numberValue)) return numberValue; + } + if (value.startsWith("{") || value.startsWith("[") || value.startsWith('"')) { + try { + const parsed = JSON.parse(value) as ASIValue; + return parsed; + } catch { + return value; + } + } + return value; +} + +export function mergeAsi(base: ASIData | null, override: ASIData | undefined): ASIData | undefined { + if (!base && !override) return undefined; + return { + ...(base ?? {}), + ...(override ?? {}), + }; +} + +export function commas(value: number): string { + const sign = value < 0 ? "-" : ""; + const digits = String(Math.trunc(Math.abs(value))); + const groups: string[] = []; + for (let index = digits.length; index > 0; index -= 3) { + groups.unshift(digits.slice(Math.max(0, index - 3), index)); + } + return sign + groups.join(","); +} + +export function fmtNum(value: number, decimals: number = 0): string { + if (decimals <= 0) return commas(Math.round(value)); + const absolute = Math.abs(value); + const whole = Math.floor(absolute); + const fraction = (absolute - whole).toFixed(decimals).slice(1); + return `${value < 0 ? "-" : ""}${commas(whole)}${fraction}`; +} + +export function formatNum(value: number | null, unit: string): string { + if (value === null) return "-"; + if (Number.isInteger(value)) return `${fmtNum(value)}${unit}`; + return `${fmtNum(value, 2)}${unit}`; +} + +export function formatElapsed(milliseconds: number): string { + const totalSeconds = Math.floor(milliseconds / 1000); + const minutes = Math.floor(totalSeconds / 60); + const seconds = totalSeconds % 60; + if (minutes > 0) { + return `${minutes}m ${String(seconds).padStart(2, "0")}s`; + } + return `${seconds}s`; +} + +export function createTempFileAllocator(): () => string { + let tempPath: string | undefined; + return () => { + if (tempPath) return tempPath; + tempPath = path.join(os.tmpdir(), `pi-autoresearch-${crypto.randomUUID()}.log`); + return tempPath; + }; +} + +export function killTree(pid: number): void { + try { + process.kill(-pid, "SIGTERM"); + } catch { + try { + process.kill(pid, "SIGTERM"); + } catch { + // Process already exited. + } + } +} + +export function isAutoresearchShCommand(command: string): boolean { + let normalized = command.trim(); + normalized = normalized.replace(/^(?:\w+=\S*\s+)+/, ""); + + let previous = ""; + while (previous !== normalized) { + previous = normalized; + normalized = normalized.replace(/^(?:env|time|nice|nohup)(?:\s+-\S+(?:\s+\d+)?)?\s+/, ""); + } + + return /^(?:(?:bash|sh)\s+(?:-\w+\s+)*)?(?:\.\/|\/[\w/.-]*\/)?autoresearch\.sh(?:\s|$)/.test(normalized); +} + +export function isBetter(current: number, best: number, direction: MetricDirection): boolean { + return direction === "lower" ? current < best : current > best; +} + +export function inferMetricUnitFromName(name: string): string { + if (name.endsWith("µs") || name.endsWith("_µs")) return "µs"; + if (name.endsWith("ms") || name.endsWith("_ms")) return "ms"; + if (name.endsWith("_s") || name.endsWith("_sec") || name.endsWith("_secs")) return "s"; + if (name.endsWith("_kb") || name.endsWith("kb")) return "kb"; + if (name.endsWith("_mb") || name.endsWith("mb")) return "mb"; + return ""; +} + +export function readConfig(cwd: string): AutoresearchConfig { + const configPath = path.join(cwd, "autoresearch.config.json"); + try { + const raw = fs.readFileSync(configPath, "utf8"); + const parsed = JSON.parse(raw) as unknown; + if (typeof parsed !== "object" || parsed === null) return {}; + const candidate = parsed as { maxIterations?: unknown; workingDir?: unknown }; + const config: AutoresearchConfig = {}; + if (typeof candidate.maxIterations === "number" && Number.isFinite(candidate.maxIterations)) { + config.maxIterations = candidate.maxIterations; + } + if (typeof candidate.workingDir === "string" && candidate.workingDir.trim().length > 0) { + config.workingDir = candidate.workingDir; + } + return config; + } catch (error) { + if (isEnoent(error)) return {}; + return {}; + } +} + +export function readMaxExperiments(cwd: string): number | null { + const value = readConfig(cwd).maxIterations; + if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return null; + return Math.floor(value); +} + +export function resolveWorkDir(cwd: string): string { + const configured = readConfig(cwd).workingDir; + if (!configured) return cwd; + return path.isAbsolute(configured) ? configured : path.resolve(cwd, configured); +} + +export function validateWorkDir(cwd: string): string | null { + const workDir = resolveWorkDir(cwd); + try { + const stat = fs.statSync(workDir); + if (!stat.isDirectory()) { + return `workingDir ${workDir} is not a directory.`; + } + return null; + } catch (error) { + if (isEnoent(error)) { + return `workingDir ${workDir} does not exist.`; + } + return `workingDir ${workDir} is unavailable.`; + } +} diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts new file mode 100644 index 000000000..126ec2701 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -0,0 +1,205 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import type { AutocompleteItem } from "@oh-my-pi/pi-tui"; +import { renderPromptTemplate } from "../config/prompt-templates"; +import type { ExtensionContext, ExtensionFactory } from "../extensibility/extensions"; +import commandStartTemplate from "./command-start.md" with { type: "text" }; +import { createDashboardController } from "./dashboard"; +import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "./helpers"; +import promptTemplate from "./prompt.md" with { type: "text" }; +import resumeMessageTemplate from "./resume-message.md" with { type: "text" }; +import { + cloneExperimentState, + createExperimentState, + createRuntimeStore, + reconstructControlState, + reconstructStateFromJsonl, +} from "./state"; +import { createInitExperimentTool } from "./tools/init-experiment"; +import { createLogExperimentTool } from "./tools/log-experiment"; +import { createRunExperimentTool } from "./tools/run-experiment"; +import type { AutoresearchRuntime } from "./types"; + +const AUTORESUME_INTERVAL_MS = 5 * 60 * 1000; +const MAX_AUTORESUME_TURNS = 20; + +export const createAutoresearchExtension: ExtensionFactory = api => { + const runtimeStore = createRuntimeStore(); + const dashboard = createDashboardController(); + + const getSessionKey = (ctx: ExtensionContext): string => ctx.sessionManager.getSessionId(); + const getRuntime = (ctx: ExtensionContext): AutoresearchRuntime => runtimeStore.ensure(getSessionKey(ctx)); + + const rehydrate = (ctx: ExtensionContext): void => { + const runtime = getRuntime(ctx); + const workDir = resolveWorkDir(ctx.cwd); + const reconstructed = reconstructStateFromJsonl(workDir); + const control = reconstructControlState(ctx.sessionManager.getEntries()); + runtime.state = cloneExperimentState(reconstructed.state); + runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); + runtime.goal = control.goal; + runtime.autoresearchMode = control.autoresearchMode; + runtime.lastAutoResumeTime = 0; + runtime.experimentsThisSession = 0; + runtime.autoResumeTurns = 0; + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.runningExperiment = null; + dashboard.updateWidget(ctx, runtime); + }; + + const setMode = ( + ctx: ExtensionContext, + enabled: boolean, + goal: string | null, + mode: "on" | "off" | "clear", + ): void => { + const runtime = getRuntime(ctx); + runtime.autoresearchMode = enabled; + runtime.goal = goal; + api.appendEntry("autoresearch-control", goal ? { mode, goal } : { mode }); + }; + + api.registerTool(createInitExperimentTool({ dashboard, getRuntime, pi: api })); + api.registerTool(createRunExperimentTool({ dashboard, getRuntime, pi: api })); + api.registerTool(createLogExperimentTool({ dashboard, getRuntime, pi: api })); + + api.registerCommand("autoresearch", { + description: "Start, stop, or clear builtin autoresearch mode.", + getArgumentCompletions(argumentPrefix: string): AutocompleteItem[] | null { + if (argumentPrefix.includes(" ")) return null; + const completions: AutocompleteItem[] = [ + { label: "off", value: "off", description: "Leave autoresearch mode" }, + { label: "clear", value: "clear", description: "Delete autoresearch.jsonl and leave autoresearch mode" }, + ]; + const normalized = argumentPrefix.trim().toLowerCase(); + const filtered = completions.filter(item => item.label.startsWith(normalized)); + return filtered.length > 0 ? filtered : null; + }, + async handler(args, ctx): Promise { + const trimmed = args.trim(); + const runtime = getRuntime(ctx); + const workDirError = validateWorkDir(ctx.cwd); + if (workDirError) { + ctx.ui.notify(workDirError, "error"); + return; + } + + if (trimmed.length === 0) { + ctx.ui.notify("Usage: /autoresearch | off | clear", "info"); + return; + } + if (trimmed === "off") { + setMode(ctx, false, runtime.goal, "off"); + runtime.experimentsThisSession = 0; + runtime.autoResumeTurns = 0; + dashboard.updateWidget(ctx, runtime); + ctx.ui.notify("Autoresearch mode disabled", "info"); + return; + } + if (trimmed === "clear") { + const workDir = resolveWorkDir(ctx.cwd); + const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + if (fs.existsSync(jsonlPath)) { + fs.rmSync(jsonlPath); + } + runtime.state = createExperimentState(); + runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); + runtime.goal = null; + setMode(ctx, false, null, "clear"); + dashboard.updateWidget(ctx, runtime); + ctx.ui.notify("Autoresearch log cleared", "info"); + return; + } + + setMode(ctx, true, trimmed, "on"); + runtime.experimentsThisSession = 0; + runtime.autoResumeTurns = 0; + dashboard.updateWidget(ctx, runtime); + api.sendUserMessage( + renderPromptTemplate(commandStartTemplate, { + goal: trimmed, + }), + ); + }, + }); + + api.registerShortcut("ctrl+x", { + description: "Toggle autoresearch dashboard", + handler(ctx): void { + const runtime = getRuntime(ctx); + if (runtime.state.results.length === 0 && !runtime.runningExperiment) { + ctx.ui.notify("No autoresearch results yet", "info"); + return; + } + runtime.dashboardExpanded = !runtime.dashboardExpanded; + dashboard.updateWidget(ctx, runtime); + }, + }); + + api.registerShortcut("ctrl+shift+x", { + description: "Show autoresearch dashboard overlay", + handler(ctx): Promise { + return dashboard.showOverlay(ctx, getRuntime(ctx)); + }, + }); + + api.on("session_start", (_event, ctx) => rehydrate(ctx)); + api.on("session_switch", (_event, ctx) => rehydrate(ctx)); + api.on("session_branch", (_event, ctx) => rehydrate(ctx)); + api.on("session_tree", (_event, ctx) => rehydrate(ctx)); + api.on("session_shutdown", (_event, ctx) => { + dashboard.clear(ctx); + runtimeStore.clear(getSessionKey(ctx)); + }); + + api.on("agent_start", (_event, ctx) => { + getRuntime(ctx).experimentsThisSession = 0; + }); + + api.on("agent_end", (_event, ctx) => { + const runtime = getRuntime(ctx); + runtime.runningExperiment = null; + dashboard.updateWidget(ctx, runtime); + dashboard.requestRender(); + if (!runtime.autoresearchMode) return; + if (runtime.experimentsThisSession === 0) return; + if (runtime.autoResumeTurns >= MAX_AUTORESUME_TURNS) return; + const now = Date.now(); + if (now - runtime.lastAutoResumeTime < AUTORESUME_INTERVAL_MS) return; + runtime.lastAutoResumeTime = now; + runtime.autoResumeTurns += 1; + const workDir = resolveWorkDir(ctx.cwd); + const ideasPath = path.join(workDir, "autoresearch.ideas.md"); + api.sendUserMessage( + renderPromptTemplate(resumeMessageTemplate, { + has_ideas: fs.existsSync(ideasPath), + }), + { deliverAs: "followUp" }, + ); + }); + + api.on("before_agent_start", (event, ctx) => { + const runtime = getRuntime(ctx); + if (!runtime.autoresearchMode) return; + const workDir = resolveWorkDir(ctx.cwd); + const autoresearchMdPath = path.join(workDir, "autoresearch.md"); + const checksPath = path.join(workDir, "autoresearch.checks.sh"); + const ideasPath = path.join(workDir, "autoresearch.ideas.md"); + return { + systemPrompt: renderPromptTemplate(promptTemplate, { + base_system_prompt: event.systemPrompt, + goal: runtime.goal ?? event.prompt, + working_dir: workDir, + default_metric_name: runtime.state.metricName, + has_autoresearch_md: fs.existsSync(autoresearchMdPath), + autoresearch_md_path: autoresearchMdPath, + has_checks: fs.existsSync(checksPath), + checks_path: checksPath, + has_ideas: fs.existsSync(ideasPath), + ideas_path: ideasPath, + }), + }; + }); +}; diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md new file mode 100644 index 000000000..56ce3efd3 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -0,0 +1,155 @@ +{{{base_system_prompt}}} + +## Autoresearch Mode + +Autoresearch mode is active. + +Primary goal: +{{goal}} + +Working directory: +`{{working_dir}}` + +You are running an autonomous experiment loop. Keep iterating until the user interrupts you or the configured maximum iteration count is reached. + +### Available tools + +- `init_experiment` — initialize or reset the experiment session for the current optimization target. +- `run_experiment` — run a benchmark or experiment command with timing, output capture, structured metric parsing, and optional backpressure checks. +- `log_experiment` — record the result, update the dashboard, persist JSONL history, auto-commit kept experiments, and auto-revert discarded or failed experiments. + +### Operating protocol + +1. Understand the target before touching code. + - Read the relevant source files. + - Identify the true bottleneck or quality constraint. + - Check existing scripts, benchmark harnesses, and config files. +2. Keep your notes in `autoresearch.md`. + - Record the goal, the benchmark command, the primary metric, important secondary metrics, and the running ideas backlog. + - Update the notes whenever the strategy changes. +3. Use `autoresearch.sh` as the canonical benchmark entrypoint. + - If it does not exist yet, create it. + - Make it print structured metric lines in the form `METRIC name=value`. + - Use the same workload every run unless you intentionally re-initialize with a new segment. +4. Initialize the loop with `init_experiment` before the first logged run of a segment. +5. Run a baseline first. + - Establish the baseline metric before attempting optimizations. + - Track secondary metrics only when they matter to correctness, quality, or obvious regressions. +6. Iterate. + - Make one coherent experiment at a time. + - Run `run_experiment`. + - Interpret the result honestly. + - Call `log_experiment` after every run. +7. Keep the primary metric as the decision maker. + - `keep` when the primary metric improves. + - `discard` when it regresses or stays flat. + - `crash` when the run fails. + - `checks_failed` when the benchmark passes but backpressure checks fail. +8. Record ASI on every `log_experiment` call. + - At minimum include `hypothesis`. + - On `discard`, `crash`, or `checks_failed`, also include `rollback_reason` and `next_action_hint`. + - Use ASI to capture what you learned, not just what you changed. +9. Prefer simpler wins. + - Remove dead ends. + - Do not keep complexity that does not move the metric. + - Do not thrash between unrelated ideas without writing down the conclusion. +10. When confidence is low, confirm. + - The dashboard confidence score compares the best observed improvement against the observed noise floor. + - Below `1.0x` usually means the improvement is within noise. + - Re-run promising changes when needed before keeping them. + +### Benchmark harness guidance + +Your benchmark script SHOULD: + +- live at `autoresearch.sh` +- run from `{{working_dir}}` +- fail with a non-zero exit status on invalid runs +- print the primary metric as `METRIC {{default_metric_name}}=` or another explicit metric name chosen during initialization +- print secondary metrics as additional `METRIC name=value` lines +- avoid extra randomness when possible +- use repeated samples and median-style summaries for fast benchmarks + +### Notes file template + +Keep `autoresearch.md` concise and current. + +Suggested structure: + +```md +# Autoresearch + +## Goal +- {{goal}} + +## Benchmark +- command: +- primary metric: +- secondary metrics: + +## Baseline +- metric: +- notes: + +## Current best +- metric: +- why it won: + +## Ideas +- item +``` + +### Guardrails + +- Do not game the benchmark. +- Do not overfit to synthetic inputs if the real workload is broader. +- Preserve correctness. +- If you create `autoresearch.checks.sh`, treat it as a hard gate for `keep`. +- If the user sends another message while a run is in progress, finish the current run and logging cycle first, then address the new input in the next iteration. + +{{#if has_autoresearch_md}} +### Resume mode + +`autoresearch.md` already exists at `{{autoresearch_md_path}}`. + +Resume from the existing notes: + +- read `autoresearch.md` +- inspect recent git history +- inspect `autoresearch.jsonl` +- continue from the most promising unfinished branch + +{{else}} +### Initial setup + +`autoresearch.md` does not exist yet. + +Create the experiment workspace before the first benchmark: + +- write `autoresearch.md` +- write `autoresearch.sh` +- optionally write `autoresearch.checks.sh` +- run `init_experiment` +- run and log the baseline + +{{/if}} +{{#if has_checks}} +### Backpressure checks + +`autoresearch.checks.sh` exists at `{{checks_path}}` and runs automatically after passing benchmark runs. + +Treat failing checks as a failed experiment: + +- do not `keep` a run when checks fail +- log it as `checks_failed` +- diagnose the regression before continuing + +{{/if}} +{{#if has_ideas}} +### Ideas backlog + +`autoresearch.ideas.md` exists at `{{ideas_path}}`. + +Use it to keep promising but deferred experiments. Prune stale ideas when they are disproven or superseded. + +{{/if}} diff --git a/packages/coding-agent/src/autoresearch/resume-message.md b/packages/coding-agent/src/autoresearch/resume-message.md new file mode 100644 index 000000000..55fda95ad --- /dev/null +++ b/packages/coding-agent/src/autoresearch/resume-message.md @@ -0,0 +1,10 @@ +The autoresearch loop ended unexpectedly. Resume it now. + +- Read `autoresearch.md` and `autoresearch.jsonl`. +- Inspect recent git history for context. +- Continue from the most promising unfinished direction. +{{#if has_ideas}} +- Review `autoresearch.ideas.md` for promising next steps and prune stale items. +{{/if}} +- Keep iterating until interrupted or until the configured iteration cap is reached. +- Preserve correctness and do not game the benchmark. diff --git a/packages/coding-agent/src/autoresearch/state.ts b/packages/coding-agent/src/autoresearch/state.ts new file mode 100644 index 000000000..9d302d9b9 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/state.ts @@ -0,0 +1,279 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import type { SessionEntry } from "../session/session-manager"; +import { inferMetricUnitFromName, isBetter } from "./helpers"; +import type { + AutoresearchControlEntryData, + AutoresearchJsonConfigEntry, + AutoresearchJsonRunEntry, + AutoresearchRuntime, + ExperimentResult, + ExperimentState, + MetricDef, + MetricDirection, + NumericMetricMap, + ReconstructedControlState, + ReconstructedExperimentData, + RuntimeStore, +} from "./types"; + +export function createExperimentState(): ExperimentState { + return { + results: [], + bestMetric: null, + bestDirection: "lower", + metricName: "metric", + metricUnit: "", + secondaryMetrics: [], + name: null, + currentSegment: 0, + maxExperiments: null, + confidence: null, + }; +} + +export function createSessionRuntime(): AutoresearchRuntime { + return { + autoresearchMode: false, + dashboardExpanded: false, + lastAutoResumeTime: 0, + experimentsThisSession: 0, + autoResumeTurns: 0, + lastRunChecks: null, + lastRunDuration: null, + lastRunAsi: null, + runningExperiment: null, + state: createExperimentState(), + goal: null, + }; +} + +export function cloneExperimentState(state: ExperimentState): ExperimentState { + return { + ...state, + results: state.results.map(result => ({ + ...result, + metrics: { ...result.metrics }, + asi: result.asi ? structuredClone(result.asi) : undefined, + })), + secondaryMetrics: state.secondaryMetrics.map(metric => ({ ...metric })), + }; +} + +export function currentResults(results: ExperimentResult[], segment: number): ExperimentResult[] { + return results.filter(result => result.segment === segment); +} + +export function findBaselineMetric(results: ExperimentResult[], segment: number): number | null { + const baseline = results.find(result => result.segment === segment); + return baseline ? baseline.metric : null; +} + +export function findBaselineRunNumber(results: ExperimentResult[], segment: number): number | null { + const index = results.findIndex(result => result.segment === segment); + return index >= 0 ? index + 1 : null; +} + +export function findBaselineSecondary( + results: ExperimentResult[], + segment: number, + knownMetrics: MetricDef[], +): NumericMetricMap { + const baseline = currentResults(results, segment)[0]; + const values: NumericMetricMap = baseline ? { ...baseline.metrics } : {}; + for (const metric of knownMetrics) { + if (values[metric.name] !== undefined) continue; + for (const result of currentResults(results, segment)) { + const value = result.metrics[metric.name]; + if (value !== undefined) { + values[metric.name] = value; + break; + } + } + } + return values; +} + +export function sortedMedian(values: number[]): number { + if (values.length === 0) return 0; + const sorted = [...values].sort((left, right) => left - right); + const midpoint = Math.floor(sorted.length / 2); + if (sorted.length % 2 === 0) { + return (sorted[midpoint - 1] + sorted[midpoint]) / 2; + } + return sorted[midpoint]; +} + +export function computeConfidence( + results: ExperimentResult[], + segment: number, + direction: MetricDirection, +): number | null { + const current = currentResults(results, segment).filter(result => result.metric > 0); + if (current.length < 3) return null; + + const values = current.map(result => result.metric); + const median = sortedMedian(values); + const mad = sortedMedian(values.map(value => Math.abs(value - median))); + if (mad === 0) return null; + + const baseline = findBaselineMetric(results, segment); + if (baseline === null) return null; + + let bestKept: number | null = null; + for (const result of current) { + if (result.status !== "keep" || result.metric <= 0) continue; + if (bestKept === null || isBetter(result.metric, bestKept, direction)) { + bestKept = result.metric; + } + } + if (bestKept === null || bestKept === baseline) return null; + + return Math.abs(bestKept - baseline) / mad; +} + +export function reconstructStateFromJsonl(workDir: string): ReconstructedExperimentData { + const state = createExperimentState(); + const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + if (!fs.existsSync(jsonlPath)) { + return { hasLog: false, state }; + } + + const content = fs.readFileSync(jsonlPath, "utf8"); + const lines = content + .split("\n") + .map(line => line.trim()) + .filter(line => line.length > 0); + + let segment = 0; + let sawConfig = false; + for (const line of lines) { + let parsed: unknown; + try { + parsed = JSON.parse(line) as unknown; + } catch { + continue; + } + + if (isConfigEntry(parsed)) { + if (sawConfig || state.results.length > 0) { + segment += 1; + } + sawConfig = true; + state.currentSegment = segment; + if (parsed.name) state.name = parsed.name; + if (parsed.metricName) state.metricName = parsed.metricName; + if (parsed.metricUnit !== undefined) state.metricUnit = parsed.metricUnit; + if (parsed.bestDirection) state.bestDirection = parsed.bestDirection; + state.secondaryMetrics = []; + continue; + } + + if (!isRunEntry(parsed)) continue; + const result: ExperimentResult = { + commit: typeof parsed.commit === "string" ? parsed.commit : "", + metric: typeof parsed.metric === "number" && Number.isFinite(parsed.metric) ? parsed.metric : 0, + metrics: cloneNumericMetrics(parsed.metrics), + status: isExperimentStatus(parsed.status) ? parsed.status : "keep", + description: typeof parsed.description === "string" ? parsed.description : "", + timestamp: typeof parsed.timestamp === "number" && Number.isFinite(parsed.timestamp) ? parsed.timestamp : 0, + segment, + confidence: + typeof parsed.confidence === "number" && Number.isFinite(parsed.confidence) ? parsed.confidence : null, + asi: cloneAsi(parsed.asi), + }; + state.results.push(result); + if (segment !== state.currentSegment) continue; + registerSecondaryMetrics(state.secondaryMetrics, result.metrics); + } + + state.bestMetric = findBaselineMetric(state.results, state.currentSegment); + state.confidence = computeConfidence(state.results, state.currentSegment, state.bestDirection); + return { hasLog: true, state }; +} + +export function reconstructControlState(entries: SessionEntry[]): ReconstructedControlState { + let autoresearchMode = false; + let goal: string | null = null; + for (const entry of entries) { + if (entry.type !== "custom" || entry.customType !== "autoresearch-control") continue; + const data = parseControlEntry(entry.data); + if (!data) continue; + autoresearchMode = data.mode === "on"; + goal = data.goal ?? goal; + if (data.mode === "clear") { + goal = null; + } + } + return { autoresearchMode, goal }; +} + +export function createRuntimeStore(): RuntimeStore { + const runtimes = new Map(); + return { + clear(sessionKey: string): void { + runtimes.delete(sessionKey); + }, + ensure(sessionKey: string): AutoresearchRuntime { + const existing = runtimes.get(sessionKey); + if (existing) return existing; + const runtime = createSessionRuntime(); + runtimes.set(sessionKey, runtime); + return runtime; + }, + }; +} + +function registerSecondaryMetrics(metrics: MetricDef[], values: NumericMetricMap): void { + for (const name of Object.keys(values)) { + if (metrics.some(metric => metric.name === name)) continue; + metrics.push({ + name, + unit: inferMetricUnitFromName(name), + }); + } +} + +function isConfigEntry(value: unknown): value is AutoresearchJsonConfigEntry { + if (typeof value !== "object" || value === null) return false; + const candidate = value as { type?: unknown }; + return candidate.type === "config"; +} + +function isRunEntry(value: unknown): value is AutoresearchJsonRunEntry { + if (typeof value !== "object" || value === null) return false; + const candidate = value as { type?: unknown }; + return candidate.type === undefined || candidate.type === "run"; +} + +function isExperimentStatus(value: unknown): value is ExperimentResult["status"] { + return value === "keep" || value === "discard" || value === "crash" || value === "checks_failed"; +} + +function cloneNumericMetrics(value: unknown): NumericMetricMap { + if (typeof value !== "object" || value === null) return {}; + const metrics = value as { [key: string]: unknown }; + const clone: NumericMetricMap = {}; + for (const [key, entryValue] of Object.entries(metrics)) { + if (typeof entryValue === "number" && Number.isFinite(entryValue)) { + clone[key] = entryValue; + } + } + return clone; +} + +function cloneAsi(value: unknown): ExperimentResult["asi"] { + if (typeof value !== "object" || value === null) return undefined; + return structuredClone(value) as ExperimentResult["asi"]; +} + +function parseControlEntry(value: unknown): AutoresearchControlEntryData | null { + if (typeof value !== "object" || value === null) return null; + const candidate = value as { goal?: unknown; mode?: unknown }; + if (candidate.mode !== "on" && candidate.mode !== "off" && candidate.mode !== "clear") return null; + const data: AutoresearchControlEntryData = { mode: candidate.mode }; + if (typeof candidate.goal === "string" && candidate.goal.trim().length > 0) { + data.goal = candidate.goal; + } + return data; +} diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts new file mode 100644 index 000000000..b6aa8a659 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -0,0 +1,117 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import { StringEnum } from "@oh-my-pi/pi-ai"; +import { Text } from "@oh-my-pi/pi-tui"; +import { Type } from "@sinclair/typebox"; +import type { ToolDefinition } from "../../extensibility/extensions"; +import type { Theme } from "../../modes/theme/theme"; +import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "../helpers"; +import { cloneExperimentState } from "../state"; +import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types"; + +const initExperimentSchema = Type.Object({ + name: Type.String({ + description: "Human-readable experiment name.", + }), + metric_name: Type.String({ + description: "Primary metric name shown in the dashboard.", + }), + metric_unit: Type.Optional( + Type.String({ + description: "Unit for the primary metric, for example µs, ms, s, kb, or empty.", + }), + ), + direction: Type.Optional( + StringEnum(["lower", "higher"], { + description: "Whether lower or higher values are better. Defaults to lower.", + }), + ), +}); + +interface InitExperimentDetails { + state: ExperimentState; +} + +export function createInitExperimentTool( + options: AutoresearchToolFactoryOptions, +): ToolDefinition { + return { + name: "init_experiment", + label: "Init Experiment", + description: + "Initialize or reset the autoresearch session for the current optimization target before the first logged run of a segment.", + parameters: initExperimentSchema, + async execute(_toolCallId, params, _signal, _onUpdate, ctx) { + const workDirError = validateWorkDir(ctx.cwd); + if (workDirError) { + return { + content: [{ type: "text", text: `Error: ${workDirError}` }], + }; + } + + const runtime = options.getRuntime(ctx); + const state = runtime.state; + const isReinitializing = state.results.length > 0; + + state.name = params.name; + state.metricName = params.metric_name; + state.metricUnit = params.metric_unit ?? ""; + state.bestDirection = params.direction ?? "lower"; + state.maxExperiments = readMaxExperiments(ctx.cwd); + state.bestMetric = null; + state.confidence = null; + state.secondaryMetrics = []; + if (isReinitializing) { + state.currentSegment += 1; + } + + const workDir = resolveWorkDir(ctx.cwd); + const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + const configLine = JSON.stringify({ + type: "config", + name: state.name, + metricName: state.metricName, + metricUnit: state.metricUnit, + bestDirection: state.bestDirection, + }); + + if (isReinitializing) { + fs.appendFileSync(jsonlPath, `${configLine}\n`); + } else { + fs.writeFileSync(jsonlPath, `${configLine}\n`); + } + + runtime.autoresearchMode = true; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + + const lines = [ + `Experiment initialized: ${state.name}`, + `Metric: ${state.metricName} (${state.metricUnit || "unitless"}, ${state.bestDirection} is better)`, + `Working directory: ${workDir}`, + isReinitializing + ? "Previous results remain in history. This starts a new segment and requires a fresh baseline." + : "Now run the baseline experiment and log it.", + ]; + if (state.maxExperiments !== null) { + lines.push(`Max iterations: ${state.maxExperiments}`); + } + + return { + content: [{ type: "text", text: lines.join("\n") }], + details: { state: cloneExperimentState(state) }, + }; + }, + renderCall(args, _options, theme): Text { + return new Text(renderInitCall(args.name, theme), 0, 0); + }, + renderResult(result): Text { + const text = result.content.find(part => part.type === "text")?.text ?? ""; + return new Text(text, 0, 0); + }, + }; +} + +function renderInitCall(name: string, theme: Theme): string { + return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", name)}`; +} diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts new file mode 100644 index 000000000..8fb002b5a --- /dev/null +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -0,0 +1,419 @@ +import * as fs from "node:fs"; +import * as path from "node:path"; +import { StringEnum } from "@oh-my-pi/pi-ai"; +import { Text } from "@oh-my-pi/pi-tui"; +import { Type } from "@sinclair/typebox"; +import type { ToolDefinition } from "../../extensibility/extensions"; +import type { Theme } from "../../modes/theme/theme"; +import { formatNum, inferMetricUnitFromName, mergeAsi, resolveWorkDir, validateWorkDir } from "../helpers"; +import { + cloneExperimentState, + computeConfidence, + currentResults, + findBaselineMetric, + findBaselineSecondary, +} from "../state"; +import type { + ASIData, + AutoresearchToolFactoryOptions, + ExperimentResult, + ExperimentState, + LogDetails, + NumericMetricMap, +} from "../types"; + +const logExperimentSchema = Type.Object({ + commit: Type.String({ + description: "Current git commit hash or placeholder.", + }), + metric: Type.Number({ + description: "Primary metric value for this run.", + }), + status: StringEnum(["keep", "discard", "crash", "checks_failed"], { + description: "Outcome for this run.", + }), + description: Type.String({ + description: "Short description of the experiment.", + }), + metrics: Type.Optional( + Type.Record(Type.String(), Type.Number(), { + description: "Secondary metrics for this run.", + }), + ), + force: Type.Optional( + Type.Boolean({ + description: "Allow introducing new secondary metrics.", + }), + ), + asi: Type.Optional( + Type.Record(Type.String(), Type.Unknown(), { + description: "Actionable side information captured for this run.", + }), + ), +}); + +const PROTECTED_AUTORESEARCH_FILES = [ + "autoresearch.jsonl", + "autoresearch.md", + "autoresearch.ideas.md", + "autoresearch.sh", + "autoresearch.checks.sh", +] as const; + +interface PreservedFile { + content: Buffer; + path: string; +} + +export function createLogExperimentTool( + options: AutoresearchToolFactoryOptions, +): ToolDefinition { + return { + name: "log_experiment", + label: "Log Experiment", + description: + "Log the experiment result, update dashboard state, persist JSONL history, and apply git keep or revert behavior.", + parameters: logExperimentSchema, + async execute(_toolCallId, params, _signal, _onUpdate, ctx) { + const workDirError = validateWorkDir(ctx.cwd); + if (workDirError) { + return { + content: [{ type: "text", text: `Error: ${workDirError}` }], + }; + } + + const runtime = options.getRuntime(ctx); + const state = runtime.state; + const workDir = resolveWorkDir(ctx.cwd); + const secondaryMetrics = cloneMetrics(params.metrics); + + if (params.status === "keep" && runtime.lastRunChecks && !runtime.lastRunChecks.pass) { + return { + content: [ + { + type: "text", + text: "Error: cannot keep this run because autoresearch.checks.sh failed. Log it as checks_failed instead.", + }, + ], + }; + } + + const validationError = validateSecondaryMetrics(state, secondaryMetrics, params.force ?? false); + if (validationError) { + return { + content: [{ type: "text", text: `Error: ${validationError}` }], + }; + } + + const mergedAsi = mergeAsi(runtime.lastRunAsi, sanitizeAsi(params.asi)); + const experiment: ExperimentResult = { + commit: params.commit.slice(0, 7), + metric: params.metric, + metrics: secondaryMetrics, + status: params.status, + description: params.description, + timestamp: Date.now(), + segment: state.currentSegment, + confidence: null, + asi: mergedAsi, + }; + + state.results.push(experiment); + runtime.experimentsThisSession += 1; + registerSecondaryMetrics(state, secondaryMetrics); + state.bestMetric = findBaselineMetric(state.results, state.currentSegment); + state.confidence = computeConfidence(state.results, state.currentSegment, state.bestDirection); + experiment.confidence = state.confidence; + + persistRun(workDir, state.results.length, experiment); + + let gitNote: string | null = null; + if (params.status === "keep") { + gitNote = await commitKeptExperiment(options, workDir, state, experiment); + } else { + gitNote = await revertFailedExperiment(options, workDir); + } + + const wallClockSeconds = runtime.lastRunDuration; + runtime.runningExperiment = null; + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + + const currentSegmentRuns = currentResults(state.results, state.currentSegment).length; + const text = buildLogText(state, experiment, currentSegmentRuns, wallClockSeconds, gitNote); + if (state.maxExperiments !== null && currentSegmentRuns >= state.maxExperiments) { + runtime.autoresearchMode = false; + } + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + + return { + content: [{ type: "text", text }], + details: { + experiment: { + ...experiment, + metrics: { ...experiment.metrics }, + asi: experiment.asi ? structuredClone(experiment.asi) : undefined, + }, + state: cloneExperimentState(state), + wallClockSeconds, + }, + }; + }, + renderCall(args, _options, theme): Text { + const color = args.status === "keep" ? "success" : args.status === "discard" ? "warning" : "error"; + return new Text( + `${theme.fg("toolTitle", theme.bold("log_experiment"))} ${theme.fg(color, args.status)} ${theme.fg("muted", args.description)}`, + 0, + 0, + ); + }, + renderResult(result, _options, theme): Text { + const details = result.details; + if (!details) { + return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + } + const summary = renderSummary(details, theme); + return new Text(summary, 0, 0); + }, + }; +} + +function cloneMetrics(value: NumericMetricMap | undefined): NumericMetricMap { + return value ? { ...value } : {}; +} + +function sanitizeAsi(value: { [key: string]: unknown } | undefined): ASIData | undefined { + if (!value) return undefined; + const result: ASIData = {}; + for (const [key, entryValue] of Object.entries(value)) { + const sanitized = sanitizeAsiValue(entryValue); + if (sanitized !== undefined) { + result[key] = sanitized; + } + } + return Object.keys(result).length > 0 ? result : undefined; +} + +function sanitizeAsiValue(value: unknown): ASIData[string] | undefined { + if (value === null) return null; + if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") return value; + if (Array.isArray(value)) { + const items = value + .map(item => sanitizeAsiValue(item)) + .filter((item): item is NonNullable => item !== undefined); + return items; + } + if (typeof value === "object") { + const objectValue = value as { [key: string]: unknown }; + const result: ASIData = {}; + for (const [key, entryValue] of Object.entries(objectValue)) { + const sanitized = sanitizeAsiValue(entryValue); + if (sanitized !== undefined) { + result[key] = sanitized; + } + } + return result; + } + return undefined; +} + +function validateSecondaryMetrics(state: ExperimentState, metrics: NumericMetricMap, force: boolean): string | null { + if (state.secondaryMetrics.length === 0) return null; + const knownNames = new Set(state.secondaryMetrics.map(metric => metric.name)); + const providedNames = new Set(Object.keys(metrics)); + + const missing = [...knownNames].filter(name => !providedNames.has(name)); + if (missing.length > 0) { + return `missing secondary metrics: ${missing.join(", ")}`; + } + + const newMetrics = [...providedNames].filter(name => !knownNames.has(name)); + if (newMetrics.length > 0 && !force) { + return `new secondary metrics require force=true: ${newMetrics.join(", ")}`; + } + return null; +} + +function registerSecondaryMetrics(state: ExperimentState, metrics: NumericMetricMap): void { + for (const name of Object.keys(metrics)) { + if (state.secondaryMetrics.some(metric => metric.name === name)) continue; + state.secondaryMetrics.push({ + name, + unit: inferMetricUnitFromName(name), + }); + } +} + +function persistRun(workDir: string, runNumber: number, experiment: ExperimentResult): void { + const entry = { + run: runNumber, + ...experiment, + }; + const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + fs.appendFileSync(jsonlPath, `${JSON.stringify(entry)}\n`); +} + +async function commitKeptExperiment( + options: AutoresearchToolFactoryOptions, + workDir: string, + state: ExperimentState, + experiment: ExperimentResult, +): Promise { + const addResult = await options.pi.exec("git", ["add", "-A"], { cwd: workDir, timeout: 10_000 }); + if (addResult.code !== 0) { + return `git add failed: ${mergeStdoutStderr(addResult).trim() || `exit ${addResult.code}`}`; + } + + const diffResult = await options.pi.exec("git", ["diff", "--cached", "--quiet"], { cwd: workDir, timeout: 10_000 }); + if (diffResult.code === 0) { + return "nothing to commit"; + } + + const payload: { [key: string]: string | number } = { + status: experiment.status, + [state.metricName]: experiment.metric, + }; + for (const [name, value] of Object.entries(experiment.metrics)) { + payload[name] = value; + } + const commitMessage = `${experiment.description}\n\nResult: ${JSON.stringify(payload)}`; + const commitResult = await options.pi.exec("git", ["commit", "-m", commitMessage], { + cwd: workDir, + timeout: 10_000, + }); + if (commitResult.code !== 0) { + return `git commit failed: ${mergeStdoutStderr(commitResult).trim() || `exit ${commitResult.code}`}`; + } + + const revParseResult = await options.pi.exec("git", ["rev-parse", "--short=7", "HEAD"], { + cwd: workDir, + timeout: 5_000, + }); + const newCommit = revParseResult.stdout.trim(); + if (newCommit.length >= 7) { + experiment.commit = newCommit; + } + const summaryLine = + mergeStdoutStderr(commitResult) + .split("\n") + .find(line => line.trim().length > 0) ?? "committed"; + return summaryLine.trim(); +} + +async function revertFailedExperiment(options: AutoresearchToolFactoryOptions, workDir: string): Promise { + const preservedFiles = preserveAutoresearchFiles(workDir); + const resetResult = await options.pi.exec("git", ["reset", "--hard", "HEAD"], { cwd: workDir, timeout: 10_000 }); + const cleanResult = await options.pi.exec("git", ["clean", "-fd"], { cwd: workDir, timeout: 10_000 }); + restoreAutoresearchFiles(preservedFiles); + + const notes: string[] = ["reverted changes"]; + if (resetResult.code !== 0) { + notes.push(`git reset failed: ${mergeStdoutStderr(resetResult).trim() || `exit ${resetResult.code}`}`); + } + if (cleanResult.code !== 0) { + notes.push(`git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`); + } + return notes.join("; "); +} + +function preserveAutoresearchFiles(workDir: string): PreservedFile[] { + const files: PreservedFile[] = []; + for (const relativePath of PROTECTED_AUTORESEARCH_FILES) { + const absolutePath = path.join(workDir, relativePath); + if (!fs.existsSync(absolutePath)) continue; + files.push({ + content: fs.readFileSync(absolutePath), + path: absolutePath, + }); + } + return files; +} + +function restoreAutoresearchFiles(files: PreservedFile[]): void { + for (const file of files) { + fs.mkdirSync(path.dirname(file.path), { recursive: true }); + fs.writeFileSync(file.path, file.content); + } +} + +function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { + return `${result.stdout}${result.stderr}`; +} + +function buildLogText( + state: ExperimentState, + experiment: ExperimentResult, + currentSegmentRuns: number, + wallClockSeconds: number | null, + gitNote: string | null, +): string { + const lines = [`Logged run #${state.results.length}: ${experiment.status} - ${experiment.description}`]; + if (wallClockSeconds !== null) { + lines.push(`Wall clock: ${wallClockSeconds.toFixed(1)}s`); + } + if (state.bestMetric !== null) { + lines.push(`Baseline ${state.metricName}: ${formatNum(state.bestMetric, state.metricUnit)}`); + } + if (currentSegmentRuns > 1 && state.bestMetric !== null && experiment.metric !== state.bestMetric) { + const delta = ((experiment.metric - state.bestMetric) / state.bestMetric) * 100; + const sign = delta > 0 ? "+" : ""; + lines.push(`This run: ${formatNum(experiment.metric, state.metricUnit)} (${sign}${delta.toFixed(1)}%)`); + } else { + lines.push(`This run: ${formatNum(experiment.metric, state.metricUnit)}`); + } + if (Object.keys(experiment.metrics).length > 0) { + const baselineSecondary = findBaselineSecondary(state.results, state.currentSegment, state.secondaryMetrics); + const parts = Object.entries(experiment.metrics).map(([name, value]) => { + const unit = state.secondaryMetrics.find(metric => metric.name === name)?.unit ?? ""; + const baseline = baselineSecondary[name]; + if (baseline === undefined || baseline === 0 || currentSegmentRuns === 1) { + return `${name}: ${formatNum(value, unit)}`; + } + const delta = ((value - baseline) / baseline) * 100; + const sign = delta > 0 ? "+" : ""; + return `${name}: ${formatNum(value, unit)} (${sign}${delta.toFixed(1)}%)`; + }); + lines.push(`Secondary metrics: ${parts.join(" ")}`); + } + if (experiment.asi) { + const asiSummary = Object.entries(experiment.asi) + .map(([key, value]) => `${key}: ${truncateAsiValue(value)}`) + .join(" | "); + lines.push(`ASI: ${asiSummary}`); + } + if (state.confidence !== null) { + const status = state.confidence >= 2 ? "likely real" : state.confidence >= 1 ? "marginal" : "within noise"; + lines.push(`Confidence: ${state.confidence.toFixed(1)}x noise floor (${status})`); + } + if (gitNote) { + lines.push(`Git: ${gitNote}`); + } + if (state.maxExperiments !== null) { + lines.push(`Progress: ${currentSegmentRuns}/${state.maxExperiments} runs in current segment`); + if (currentSegmentRuns >= state.maxExperiments) { + lines.push(`Maximum experiments reached (${state.maxExperiments}). Autoresearch mode is now off.`); + } + } + return lines.join("\n"); +} + +function truncateAsiValue(value: ASIData[string]): string { + const text = typeof value === "string" ? value : JSON.stringify(value); + return text.length > 120 ? `${text.slice(0, 117)}...` : text; +} + +function renderSummary(details: LogDetails, theme: Theme): string { + const { experiment, state } = details; + const color = experiment.status === "keep" ? "success" : experiment.status === "discard" ? "warning" : "error"; + let summary = `${theme.fg(color, experiment.status.toUpperCase())} ${theme.fg("muted", experiment.description)}`; + summary += ` ${theme.fg("accent", `${state.metricName}=${formatNum(experiment.metric, state.metricUnit)}`)}`; + if (state.bestMetric !== null) { + summary += ` ${theme.fg("dim", `baseline ${formatNum(state.bestMetric, state.metricUnit)}`)}`; + } + if (state.confidence !== null) { + summary += ` ${theme.fg("dim", `conf ${state.confidence.toFixed(1)}x`)}`; + } + return summary; +} diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts new file mode 100644 index 000000000..993a6b2d3 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -0,0 +1,478 @@ +import * as childProcess from "node:child_process"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import { Text } from "@oh-my-pi/pi-tui"; +import { formatBytes } from "@oh-my-pi/pi-utils"; +import { Type } from "@sinclair/typebox"; +import type { ToolDefinition } from "../../extensibility/extensions"; +import type { Theme } from "../../modes/theme/theme"; +import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output"; +import { + createTempFileAllocator, + EXPERIMENT_MAX_BYTES, + EXPERIMENT_MAX_LINES, + formatElapsed, + formatNum, + isAutoresearchShCommand, + killTree, + parseAsiLines, + parseMetricLines, + resolveWorkDir, + validateWorkDir, +} from "../helpers"; +import type { AutoresearchToolFactoryOptions, RunDetails, RunExperimentProgressDetails } from "../types"; + +const runExperimentSchema = Type.Object({ + command: Type.String({ + description: "Shell command to run for this experiment.", + }), + timeout_seconds: Type.Optional( + Type.Number({ + description: "Timeout in seconds. Defaults to 600.", + }), + ), + checks_timeout_seconds: Type.Optional( + Type.Number({ + description: "Timeout in seconds for autoresearch.checks.sh. Defaults to 300.", + }), + ), +}); + +interface ProcessExecutionResult { + actualTotalBytes: number; + exitCode: number | null; + killed: boolean; + output: string; + tempFilePath?: string; +} + +interface ChecksExecutionResult { + code: number | null; + killed: boolean; + output: string; +} + +export function createRunExperimentTool( + options: AutoresearchToolFactoryOptions, +): ToolDefinition { + return { + name: "run_experiment", + label: "Run Experiment", + description: + "Run an experiment command with timing, tail capture, structured metric parsing, and optional autoresearch.checks.sh validation.", + parameters: runExperimentSchema, + async execute(_toolCallId, params, signal, onUpdate, ctx) { + const workDirError = validateWorkDir(ctx.cwd); + if (workDirError) { + return { + content: [{ type: "text", text: `Error: ${workDirError}` }], + }; + } + + const runtime = options.getRuntime(ctx); + const state = runtime.state; + const workDir = resolveWorkDir(ctx.cwd); + const checksPath = path.join(workDir, "autoresearch.checks.sh"); + const autoresearchScriptPath = path.join(workDir, "autoresearch.sh"); + + if (fs.existsSync(autoresearchScriptPath) && !isAutoresearchShCommand(params.command)) { + return { + content: [ + { + type: "text", + text: + `Error: autoresearch.sh exists. Run it directly instead of using a different command.\n` + + `Expected something like: bash autoresearch.sh\n` + + `Received: ${params.command}`, + }, + ], + }; + } + + if (state.maxExperiments !== null) { + const segmentRuns = state.results.filter(result => result.segment === state.currentSegment).length; + if (segmentRuns >= state.maxExperiments) { + return { + content: [ + { + type: "text", + text: `Maximum experiments reached (${state.maxExperiments}). Re-initialize to start a new segment.`, + }, + ], + }; + } + } + + runtime.runningExperiment = { + startedAt: Date.now(), + command: params.command, + }; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + + const timeoutMs = Math.max(0, Math.floor((params.timeout_seconds ?? 600) * 1000)); + const startedAt = Date.now(); + let execution: ProcessExecutionResult; + try { + execution = await executeProcess({ + command: params.command, + cwd: workDir, + timeoutMs, + signal, + onProgress: details => { + onUpdate?.({ + content: [{ type: "text", text: details.tailOutput }], + details: { + phase: "running", + elapsed: details.elapsed, + truncation: details.truncation, + fullOutputPath: details.fullOutputPath, + }, + }); + }, + }); + } finally { + runtime.runningExperiment = null; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + } + + const durationSeconds = (Date.now() - startedAt) / 1000; + runtime.lastRunDuration = durationSeconds; + + const benchmarkPassed = execution.exitCode === 0 && !execution.killed; + let checksPass: boolean | null = null; + let checksTimedOut = false; + let checksOutput = ""; + let checksDuration = 0; + + if (benchmarkPassed && fs.existsSync(checksPath)) { + const checksStartedAt = Date.now(); + const checksResult = runChecks({ + cwd: workDir, + pathToChecks: checksPath, + timeoutMs: Math.max(0, Math.floor((params.checks_timeout_seconds ?? 300) * 1000)), + signal, + }); + checksDuration = (Date.now() - checksStartedAt) / 1000; + checksTimedOut = checksResult.killed; + checksPass = checksResult.code === 0 && !checksResult.killed; + checksOutput = checksResult.output; + } + + runtime.lastRunChecks = + checksPass === null + ? null + : { + pass: checksPass, + output: checksOutput, + duration: checksDuration, + }; + + const llmTruncation = truncateTail(execution.output, { + maxBytes: EXPERIMENT_MAX_BYTES, + maxLines: EXPERIMENT_MAX_LINES, + }); + const displayTruncation = truncateTail(execution.output, { + maxBytes: DEFAULT_MAX_BYTES, + maxLines: DEFAULT_MAX_LINES, + }); + + let fullOutputPath = execution.tempFilePath; + if (!fullOutputPath && llmTruncation.truncated) { + fullOutputPath = createTempFileAllocator()(); + fs.writeFileSync(fullOutputPath, execution.output); + } + + const parsedMetricsMap = parseMetricLines(execution.output); + const parsedMetrics = parsedMetricsMap.size > 0 ? Object.fromEntries(parsedMetricsMap.entries()) : null; + const parsedPrimary = parsedMetricsMap.get(state.metricName) ?? null; + const parsedAsi = parseAsiLines(execution.output); + runtime.lastRunAsi = parsedAsi; + + const resultDetails: RunDetails = { + command: params.command, + exitCode: execution.exitCode, + durationSeconds, + passed: benchmarkPassed && (checksPass === null || checksPass), + crashed: execution.exitCode !== 0 || execution.killed || checksPass === false, + timedOut: execution.killed, + tailOutput: displayTruncation.content, + checksPass, + checksTimedOut, + checksOutput: checksOutput.split("\n").slice(-80).join("\n"), + checksDuration, + parsedMetrics, + parsedPrimary, + parsedAsi, + metricName: state.metricName, + metricUnit: state.metricUnit, + truncation: llmTruncation.truncated ? llmTruncation : undefined, + fullOutputPath, + }; + + return { + content: [{ type: "text", text: buildRunText(resultDetails, llmTruncation.content, state.bestMetric) }], + details: resultDetails, + }; + }, + renderCall(args, _options, theme): Text { + return new Text( + `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", args.command)}`, + 0, + 0, + ); + }, + renderResult(result, options, theme): Text { + if (isProgressDetails(result.details)) { + const header = theme.fg("warning", `Running ${result.details.elapsed}...`); + const preview = result.content.find(part => part.type === "text")?.text ?? ""; + return new Text(preview ? `${header}\n${theme.fg("dim", preview)}` : header, 0, 0); + } + + const details = result.details; + if (!details || !isRunDetails(details)) { + return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + } + + const statusText = renderStatus(details, theme); + if (!options.expanded && details.tailOutput.trim().length === 0) { + return new Text(statusText, 0, 0); + } + + const preview = options.expanded ? details.tailOutput : details.tailOutput.split("\n").slice(-5).join("\n"); + const suffix = + options.expanded && details.truncation && details.fullOutputPath + ? `\n${theme.fg("warning", `Full output: ${details.fullOutputPath}`)}` + : ""; + return new Text(preview ? `${statusText}\n${theme.fg("dim", preview)}${suffix}` : statusText, 0, 0); + }, + }; +} + +interface ProgressSnapshot { + elapsed: string; + fullOutputPath?: string; + tailOutput: string; + truncation?: RunExperimentProgressDetails["truncation"]; +} + +async function executeProcess(options: { + command: string; + cwd: string; + timeoutMs: number; + signal?: AbortSignal; + onProgress(details: ProgressSnapshot): void; +}): Promise { + const { promise, resolve, reject } = Promise.withResolvers(); + const child = childProcess.spawn("bash", ["-lc", options.command], { + cwd: options.cwd, + detached: true, + stdio: ["ignore", "pipe", "pipe"], + }); + + const getTempFile = createTempFileAllocator(); + const chunks: Buffer[] = []; + let chunksBytes = 0; + let totalBytes = 0; + let killedByTimeout = false; + let resolved = false; + let fullOutputPath: string | undefined; + let writeStream: fs.WriteStream | undefined; + + const cleanup = (): void => { + if (progressTimer) clearInterval(progressTimer); + if (timeoutHandle) clearTimeout(timeoutHandle); + options.signal?.removeEventListener("abort", abortHandler); + if (writeStream) { + writeStream.end(); + writeStream = undefined; + } + }; + + const finish = (callback: () => void): void => { + if (resolved) return; + resolved = true; + cleanup(); + callback(); + }; + + const appendChunk = (data: Buffer): void => { + totalBytes += data.length; + if (!fullOutputPath && totalBytes > DEFAULT_MAX_BYTES) { + fullOutputPath = getTempFile(); + writeStream = fs.createWriteStream(fullOutputPath); + for (const chunk of chunks) { + writeStream.write(chunk); + } + } + writeStream?.write(data); + chunks.push(data); + chunksBytes += data.length; + while (chunksBytes > DEFAULT_MAX_BYTES * 2 && chunks.length > 1) { + const removed = chunks.shift(); + if (removed) chunksBytes -= removed.length; + } + }; + + const snapshot = (): ProgressSnapshot => { + const tail = truncateTail(Buffer.concat(chunks).toString("utf8"), { + maxBytes: DEFAULT_MAX_BYTES, + maxLines: DEFAULT_MAX_LINES, + }); + return { + elapsed: formatElapsed(Date.now() - startedAt), + fullOutputPath, + tailOutput: tail.content, + truncation: tail.truncated ? tail : undefined, + }; + }; + + const startedAt = Date.now(); + const progressTimer = setInterval(() => { + options.onProgress(snapshot()); + }, 1000); + const timeoutHandle = + options.timeoutMs > 0 + ? setTimeout(() => { + killedByTimeout = true; + if (child.pid) killTree(child.pid); + }, options.timeoutMs) + : undefined; + + const abortHandler = (): void => { + if (child.pid) killTree(child.pid); + }; + if (options.signal?.aborted) { + abortHandler(); + } else { + options.signal?.addEventListener("abort", abortHandler, { once: true }); + } + + child.stdout?.on("data", data => { + appendChunk(data); + }); + child.stderr?.on("data", data => { + appendChunk(data); + }); + child.on("error", error => { + finish(() => reject(error)); + }); + child.on("close", code => { + if (options.signal?.aborted) { + finish(() => reject(new Error("aborted"))); + return; + } + const output = Buffer.concat(chunks).toString("utf8"); + finish(() => + resolve({ + actualTotalBytes: totalBytes, + exitCode: code, + killed: killedByTimeout, + output, + tempFilePath: fullOutputPath, + }), + ); + }); + + return promise; +} + +function runChecks(options: { + cwd: string; + pathToChecks: string; + timeoutMs: number; + signal?: AbortSignal; + // signal currently unused because spawnSync does not support AbortSignal directly. +}): ChecksExecutionResult { + const result = childProcess.spawnSync("bash", [options.pathToChecks], { + cwd: options.cwd, + timeout: options.timeoutMs, + encoding: "utf8", + maxBuffer: DEFAULT_MAX_BYTES, + }); + return { + code: result.status, + killed: result.signal === "SIGTERM" || result.signal === "SIGKILL" || Boolean(result.error), + output: `${result.stdout ?? ""}${result.stderr ?? ""}`.trim(), + }; +} + +function buildRunText(details: RunDetails, outputPreview: string, bestMetric: number | null): string { + const lines: string[] = []; + if (details.timedOut) { + lines.push(`TIMEOUT after ${details.durationSeconds.toFixed(1)}s`); + } else if (details.exitCode !== 0) { + lines.push(`FAILED with exit code ${details.exitCode} in ${details.durationSeconds.toFixed(1)}s`); + } else { + lines.push(`PASSED in ${details.durationSeconds.toFixed(1)}s`); + } + if (details.checksTimedOut) { + lines.push(`Checks timed out after ${details.checksDuration.toFixed(1)}s`); + } else if (details.checksPass === false) { + lines.push(`Checks failed in ${details.checksDuration.toFixed(1)}s`); + } else if (details.checksPass === true) { + lines.push(`Checks passed in ${details.checksDuration.toFixed(1)}s`); + } + if (bestMetric !== null) { + lines.push(`Current baseline ${details.metricName}: ${formatNum(bestMetric, details.metricUnit)}`); + } + if (details.parsedPrimary !== null) { + lines.push(`Parsed ${details.metricName}: ${details.parsedPrimary}`); + } + if (details.parsedMetrics) { + const secondary = Object.entries(details.parsedMetrics) + .filter(([name]) => name !== details.metricName) + .map(([name, value]) => `${name}=${value}`); + if (secondary.length > 0) { + lines.push(`Parsed metrics: ${secondary.join(", ")}`); + } + } + if (details.parsedAsi) { + lines.push(`Parsed ASI keys: ${Object.keys(details.parsedAsi).join(", ")}`); + } + lines.push(""); + lines.push(outputPreview); + if (details.truncation && details.fullOutputPath) { + lines.push(""); + lines.push( + `Output truncated (${formatBytes(EXPERIMENT_MAX_BYTES)} limit). Full output: ${details.fullOutputPath}`, + ); + } + if (details.checksPass === false && details.checksOutput.length > 0) { + lines.push(""); + lines.push("Checks output:"); + lines.push(details.checksOutput); + } + return lines.join("\n").trimEnd(); +} + +function renderStatus(details: RunDetails, theme: Theme): string { + if (details.timedOut) { + return theme.fg("error", `TIMEOUT ${details.durationSeconds.toFixed(1)}s`); + } + if (details.checksTimedOut) { + return theme.fg("warning", `Checks timeout ${details.checksDuration.toFixed(1)}s`); + } + if (details.checksPass === false) { + return theme.fg("error", `Checks failed ${details.checksDuration.toFixed(1)}s`); + } + if (details.exitCode !== 0) { + return theme.fg("error", `FAIL exit=${details.exitCode} ${details.durationSeconds.toFixed(1)}s`); + } + const metric = + details.parsedPrimary !== null + ? ` ${details.metricName}=${formatNum(details.parsedPrimary, details.metricUnit)}` + : ""; + return theme.fg("success", `PASS ${details.durationSeconds.toFixed(1)}s${metric}`); +} + +function isRunDetails(value: unknown): value is RunDetails { + if (typeof value !== "object" || value === null) return false; + return "command" in value && "durationSeconds" in value; +} + +function isProgressDetails(value: unknown): value is RunExperimentProgressDetails { + if (typeof value !== "object" || value === null) return false; + return "phase" in value && value.phase === "running"; +} diff --git a/packages/coding-agent/src/autoresearch/types.ts b/packages/coding-agent/src/autoresearch/types.ts new file mode 100644 index 000000000..52f7b89ec --- /dev/null +++ b/packages/coding-agent/src/autoresearch/types.ts @@ -0,0 +1,167 @@ +import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { ExtensionAPI, ExtensionContext } from "../extensibility/extensions"; +import type { SessionEntry } from "../session/session-manager"; +import type { TruncationResult } from "../session/streaming-output"; + +export type MetricDirection = "lower" | "higher"; +export type ExperimentStatus = "keep" | "discard" | "crash" | "checks_failed"; + +export type ASIValue = string | number | boolean | null | ASIValue[] | { [key: string]: ASIValue }; + +export interface ASIData { + [key: string]: ASIValue; +} + +export interface NumericMetricMap { + [key: string]: number; +} + +export interface MetricDef { + name: string; + unit: string; +} + +export interface ExperimentResult { + commit: string; + metric: number; + metrics: NumericMetricMap; + status: ExperimentStatus; + description: string; + timestamp: number; + segment: number; + confidence: number | null; + asi?: ASIData; +} + +export interface ExperimentState { + results: ExperimentResult[]; + bestMetric: number | null; + bestDirection: MetricDirection; + metricName: string; + metricUnit: string; + secondaryMetrics: MetricDef[]; + name: string | null; + currentSegment: number; + maxExperiments: number | null; + confidence: number | null; +} + +export interface RunExperimentProgressDetails { + phase: "running"; + elapsed: string; + truncation?: TruncationResult; + fullOutputPath?: string; +} + +export interface RunDetails { + command: string; + exitCode: number | null; + durationSeconds: number; + passed: boolean; + crashed: boolean; + timedOut: boolean; + tailOutput: string; + checksPass: boolean | null; + checksTimedOut: boolean; + checksOutput: string; + checksDuration: number; + parsedMetrics: NumericMetricMap | null; + parsedPrimary: number | null; + parsedAsi: ASIData | null; + metricName: string; + metricUnit: string; + truncation?: TruncationResult; + fullOutputPath?: string; +} + +export interface LogDetails { + experiment: ExperimentResult; + state: ExperimentState; + wallClockSeconds: number | null; +} + +export interface ChecksResult { + pass: boolean; + output: string; + duration: number; +} + +export interface RunningExperiment { + startedAt: number; + command: string; +} + +export interface AutoresearchRuntime { + autoresearchMode: boolean; + dashboardExpanded: boolean; + lastAutoResumeTime: number; + experimentsThisSession: number; + autoResumeTurns: number; + lastRunChecks: ChecksResult | null; + lastRunDuration: number | null; + lastRunAsi: ASIData | null; + runningExperiment: RunningExperiment | null; + state: ExperimentState; + goal: string | null; +} + +export interface AutoresearchConfig { + maxIterations?: number; + workingDir?: string; +} + +export interface AutoresearchJsonConfigEntry { + type: "config"; + name?: string; + metricName?: string; + metricUnit?: string; + bestDirection?: MetricDirection; +} + +export interface AutoresearchJsonRunEntry { + run?: number; + commit?: string; + metric?: number; + metrics?: NumericMetricMap; + status?: ExperimentStatus; + description?: string; + timestamp?: number; + confidence?: number | null; + asi?: ASIData; +} + +export interface ReconstructedExperimentData { + hasLog: boolean; + state: ExperimentState; +} + +export interface AutoresearchControlEntryData { + mode: "on" | "off" | "clear"; + goal?: string; +} + +export interface ReconstructedControlState { + autoresearchMode: boolean; + goal: string | null; +} + +export interface RuntimeStore { + clear(sessionKey: string): void; + ensure(sessionKey: string): AutoresearchRuntime; +} + +export interface DashboardController { + clear(ctx: ExtensionContext): void; + requestRender(): void; + showOverlay(ctx: ExtensionContext, runtime: AutoresearchRuntime): Promise; + updateWidget(ctx: ExtensionContext, runtime: AutoresearchRuntime): void; +} + +export interface AutoresearchToolFactoryOptions { + dashboard: DashboardController; + getRuntime(ctx: ExtensionContext): AutoresearchRuntime; + pi: ExtensionAPI; +} + +export type AutoresearchToolResult = AgentToolResult; +export type SessionEntries = SessionEntry[]; diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 418c0c37f..035597f64 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -87,6 +87,16 @@ export interface ExtensionUIDialogOptions { /** Raw terminal input listener for extensions. */ export type TerminalInputHandler = (data: string) => { consume?: boolean; data?: string } | undefined; +export type WidgetPlacement = "aboveEditor" | "belowEditor"; + +export interface ExtensionWidgetOptions { + placement?: WidgetPlacement; +} + +export type ExtensionUiComponent = Component & { dispose?(): void }; +export type ExtensionUiComponentFactory = (tui: TUI, theme: Theme) => ExtensionUiComponent; +export type ExtensionWidgetContent = string[] | ExtensionUiComponentFactory | undefined; + /** * UI context for extensions to request interactive UI. * Each mode (interactive, RPC, print) provides its own implementation. @@ -113,15 +123,14 @@ export interface ExtensionUIContext { /** Set the working/loading message shown during streaming. Call with no argument to restore default. */ setWorkingMessage(message?: string): void; - /** Set a widget to display above the editor. Accepts string array or component factory. */ - setWidget(key: string, content: string[] | undefined): void; - setWidget(key: string, content: ((tui: TUI, theme: Theme) => Component & { dispose?(): void }) | undefined): void; + /** Set a widget to display above or below the editor. Accepts string array or component factory. */ + setWidget(key: string, content: ExtensionWidgetContent, options?: ExtensionWidgetOptions): void; /** Set a custom footer component, or undefined to restore the built-in footer. */ - setFooter(factory: ((tui: TUI, theme: Theme) => Component & { dispose?(): void }) | undefined): void; + setFooter(factory: ExtensionUiComponentFactory | undefined): void; /** Set a custom header component, or undefined to restore the built-in header. */ - setHeader(factory: ((tui: TUI, theme: Theme) => Component & { dispose?(): void }) | undefined): void; + setHeader(factory: ExtensionUiComponentFactory | undefined): void; /** Set the terminal window/tab title. */ setTitle(title: string): void; @@ -133,7 +142,7 @@ export interface ExtensionUIContext { theme: Theme, keybindings: KeybindingsManager, done: (result: T) => void, - ) => (Component & { dispose?(): void }) | Promise, + ) => ExtensionUiComponent | Promise, options?: { overlay?: boolean }, ): Promise; diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index f3f911f95..5e242680b 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -1,5 +1,5 @@ import type { Component, OverlayHandle, TUI } from "@oh-my-pi/pi-tui"; -import { Spacer, Text } from "@oh-my-pi/pi-tui"; +import { Container, Spacer, Text } from "@oh-my-pi/pi-tui"; import { logger } from "@oh-my-pi/pi-utils"; import { KeybindingsManager } from "../../config/keybindings"; import type { @@ -9,6 +9,9 @@ import type { ExtensionError, ExtensionUIContext, ExtensionUIDialogOptions, + ExtensionUiComponent, + ExtensionWidgetContent, + ExtensionWidgetOptions, TerminalInputHandler, } from "../../extensibility/extensions"; import { HookEditorComponent } from "../../modes/components/hook-editor"; @@ -18,8 +21,12 @@ import { getAvailableThemesWithPaths, getThemeByName, setTheme, type Theme, them import type { InteractiveModeContext } from "../../modes/types"; import { setSessionTerminalTitle, setTerminalTitle } from "../../utils/title-generator"; +const MAX_WIDGET_LINES = 10; + export class ExtensionUiController { #extensionTerminalInputUnsubscribers = new Set<() => void>(); + #hookWidgetsAbove = new Map(); + #hookWidgetsBelow = new Map(); constructor(private ctx: InteractiveModeContext) {} /** @@ -35,7 +42,7 @@ export class ExtensionUiController { onTerminalInput: handler => this.addExtensionTerminalInputListener(handler), setStatus: (key, text) => this.setHookStatus(key, text), setWorkingMessage: message => this.ctx.setWorkingMessage(message), - setWidget: (key, content) => this.setHookWidget(key, content), + setWidget: (key, content, options) => this.setHookWidget(key, content, options), setTitle: title => setTerminalTitle(title), custom: (factory, options) => this.showHookCustom(factory, options), setEditorText: text => this.ctx.editor.setText(text), @@ -150,6 +157,7 @@ export class ExtensionUiController { // Create new session this.clearExtensionTerminalInputListeners(); + this.clearHookWidgets(); const success = await this.ctx.session.newSession({ parentSession: options?.parentSession }); if (!success) { return { cancelled: true }; @@ -227,6 +235,7 @@ export class ExtensionUiController { await this.ctx.executeCompaction(instructionsOrOptions, false); }, switchSession: async sessionPath => { + this.clearHookWidgets(); const result = await this.ctx.session.switchSession(sessionPath); if (!result) { return { cancelled: true }; @@ -252,11 +261,73 @@ export class ExtensionUiController { }); } - setHookWidget(key: string, content: unknown): void { - this.ctx.statusLine.setHookStatus(key, content === undefined || content === null ? undefined : String(content)); + setHookWidget(key: string, content: ExtensionWidgetContent, options?: ExtensionWidgetOptions): void { + const placement = options?.placement ?? "aboveEditor"; + this.#removeHookWidget(this.#hookWidgetsAbove, key); + this.#removeHookWidget(this.#hookWidgetsBelow, key); + + if (content === undefined) { + this.#rebuildHookWidgets(); + return; + } + + const target = placement === "belowEditor" ? this.#hookWidgetsBelow : this.#hookWidgetsAbove; + target.set(key, this.#createHookWidget(content)); + this.#rebuildHookWidgets(); + } + + #removeHookWidget(widgets: Map, key: string): void { + const existing = widgets.get(key); + existing?.dispose?.(); + widgets.delete(key); + } + + #createHookWidget(content: ExtensionWidgetContent): ExtensionUiComponent { + if (Array.isArray(content)) { + const container = new Container(); + for (const line of content.slice(0, MAX_WIDGET_LINES)) { + container.addChild(new Text(line, 1, 0)); + } + if (content.length > MAX_WIDGET_LINES) { + container.addChild(new Text(theme.fg("muted", "... (widget truncated)"), 1, 0)); + } + return container; + } + if (content === undefined) { + throw new Error("Widget content missing"); + } + return content(this.ctx.ui, theme); + } + + #rebuildHookWidgets(): void { + this.#renderHookWidgetContainer(this.ctx.hookWidgetContainerAbove, this.#hookWidgetsAbove, true, true); + this.#renderHookWidgetContainer(this.ctx.hookWidgetContainerBelow, this.#hookWidgetsBelow, false, false); this.ctx.ui.requestRender(); } + #renderHookWidgetContainer( + container: Container, + widgets: Map, + spacerWhenEmpty: boolean, + leadingSpacer: boolean, + ): void { + container.clear(); + + if (widgets.size === 0) { + if (spacerWhenEmpty) { + container.addChild(new Spacer(1)); + } + return; + } + + if (leadingSpacer) { + container.addChild(new Spacer(1)); + } + for (const widget of widgets.values()) { + container.addChild(widget); + } + } + initializeHookRunner(uiContext: ExtensionUIContext, _hasUI: boolean): void { const extensionRunner = this.ctx.session.extensionRunner; if (!extensionRunner) { @@ -353,6 +424,7 @@ export class ExtensionUiController { // Create new session this.clearExtensionTerminalInputListeners(); + this.clearHookWidgets(); const success = await this.ctx.session.newSession({ parentSession: options?.parentSession }); if (!success) { return { cancelled: true }; @@ -432,6 +504,7 @@ export class ExtensionUiController { if (this.ctx.isBackgrounded) { return { cancelled: true }; } + this.clearHookWidgets(); const result = await this.ctx.session.switchSession(sessionPath); if (!result) { return { cancelled: true }; @@ -828,6 +901,18 @@ export class ExtensionUiController { }; } + clearHookWidgets(): void { + for (const widget of this.#hookWidgetsAbove.values()) { + widget.dispose?.(); + } + for (const widget of this.#hookWidgetsBelow.values()) { + widget.dispose?.(); + } + this.#hookWidgetsAbove.clear(); + this.#hookWidgetsBelow.clear(); + this.#rebuildHookWidgets(); + } + clearExtensionTerminalInputListeners(): void { for (const unsubscribe of this.#extensionTerminalInputUnsubscribers) { unsubscribe(); diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index ab129bf1b..ab0c53494 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -12,7 +12,12 @@ import chalk from "chalk"; import { KeybindingsManager } from "../config/keybindings"; import { renderPromptTemplate } from "../config/prompt-templates"; import { type Settings, settings } from "../config/settings"; -import type { ExtensionUIContext, ExtensionUIDialogOptions } from "../extensibility/extensions"; +import type { + ExtensionUIContext, + ExtensionUIDialogOptions, + ExtensionWidgetContent, + ExtensionWidgetOptions, +} from "../extensibility/extensions"; import type { CompactOptions } from "../extensibility/extensions/types"; import { BUILTIN_SLASH_COMMANDS, loadSlashCommands } from "../extensibility/slash-commands"; import { resolveLocalUrlToPath } from "../internal-urls"; @@ -93,6 +98,8 @@ export class InteractiveMode implements InteractiveModeContext { btwContainer: Container; editor: CustomEditor; editorContainer: Container; + hookWidgetContainerAbove: Container; + hookWidgetContainerBelow: Container; statusLine: StatusLineComponent; isInitialized = false; @@ -216,6 +223,9 @@ export class InteractiveMode implements InteractiveModeContext { } catch (error) { logger.warn("History storage unavailable", { error: String(error) }); } + this.hookWidgetContainerAbove = new Container(); + this.hookWidgetContainerAbove.addChild(new Spacer(1)); + this.hookWidgetContainerBelow = new Container(); this.editorContainer = new Container(); this.editorContainer.addChild(this.editor); this.statusLine = new StatusLineComponent(session); @@ -329,8 +339,9 @@ export class InteractiveMode implements InteractiveModeContext { this.ui.addChild(this.todoContainer); this.ui.addChild(this.btwContainer); this.ui.addChild(this.statusLine); // Only renders hook statuses (main status in editor border) - this.ui.addChild(new Spacer(1)); + this.ui.addChild(this.hookWidgetContainerAbove); this.ui.addChild(this.editorContainer); + this.ui.addChild(this.hookWidgetContainerBelow); this.ui.setFocus(this.editor); this.#inputController.setupKeyHandlers(); @@ -837,6 +848,7 @@ export class InteractiveMode implements InteractiveModeContext { this.#sttController = undefined; } this.#extensionUiController.clearExtensionTerminalInputListeners(); + this.#extensionUiController.clearHookWidgets(); this.statusLine.dispose(); if (this.#resizeHandler) { process.stdout.removeListener("resize", this.#resizeHandler); @@ -1359,8 +1371,8 @@ export class InteractiveMode implements InteractiveModeContext { return this.#extensionUiController.emitCustomToolSessionEvent(reason, previousSessionFile); } - setHookWidget(key: string, content: unknown): void { - this.#extensionUiController.setHookWidget(key, content); + setHookWidget(key: string, content: ExtensionWidgetContent, options?: ExtensionWidgetOptions): void { + this.#extensionUiController.setHookWidget(key, content, options); } setHookStatus(key: string, text: string | undefined): void { diff --git a/packages/coding-agent/src/modes/rpc/rpc-mode.ts b/packages/coding-agent/src/modes/rpc/rpc-mode.ts index 1e1642968..ff5f95818 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-mode.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-mode.ts @@ -11,7 +11,11 @@ * - Extension UI: Extension UI requests are emitted, client responds with extension_ui_response */ import { readJsonl, Snowflake } from "@oh-my-pi/pi-utils"; -import type { ExtensionUIContext, ExtensionUIDialogOptions } from "../../extensibility/extensions"; +import type { + ExtensionUIContext, + ExtensionUIDialogOptions, + ExtensionWidgetOptions, +} from "../../extensibility/extensions"; import { type Theme, theme } from "../../modes/theme/theme"; import type { AgentSession } from "../../session/agent-session"; import type { @@ -198,7 +202,7 @@ export async function runRpcMode(session: AgentSession): Promise { // Not supported in RPC mode } - setWidget(key: string, content: unknown): void { + setWidget(key: string, content: unknown, options?: ExtensionWidgetOptions): void { // Only support string arrays in RPC mode - factory functions are ignored if (content === undefined || Array.isArray(content)) { this.output({ @@ -207,6 +211,7 @@ export async function runRpcMode(session: AgentSession): Promise { method: "setWidget", widgetKey: key, widgetLines: content as string[] | undefined, + widgetPlacement: options?.placement, } as RpcExtensionUIRequest); } // Component factories are not supported in RPC mode - would need TUI access diff --git a/packages/coding-agent/src/modes/rpc/rpc-types.ts b/packages/coding-agent/src/modes/rpc/rpc-types.ts index d212d70d7..558d58ea4 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-types.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-types.ts @@ -215,6 +215,7 @@ export type RpcExtensionUIRequest = method: "setWidget"; widgetKey: string; widgetLines: string[] | undefined; + widgetPlacement?: "aboveEditor" | "belowEditor"; } | { type: "extension_ui_request"; id: string; method: "setTitle"; title: string } | { type: "extension_ui_request"; id: string; method: "set_editor_text"; text: string }; diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 705020715..bd7fd1766 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -3,7 +3,12 @@ import type { AssistantMessage, ImageContent, Message, UsageReport } from "@oh-m import type { Component, Container, Loader, Spacer, Text, TUI } from "@oh-my-pi/pi-tui"; import type { KeybindingsManager } from "../config/keybindings"; import type { Settings } from "../config/settings"; -import type { ExtensionUIContext, ExtensionUIDialogOptions } from "../extensibility/extensions"; +import type { + ExtensionUIContext, + ExtensionUIDialogOptions, + ExtensionWidgetContent, + ExtensionWidgetOptions, +} from "../extensibility/extensions"; import type { CompactOptions } from "../extensibility/extensions/types"; import type { MCPManager } from "../mcp"; import type { AgentSession, AgentSessionEvent } from "../session/agent-session"; @@ -59,6 +64,8 @@ export interface InteractiveModeContext { btwContainer: Container; editor: CustomEditor; editorContainer: Container; + hookWidgetContainerAbove: Container; + hookWidgetContainerBelow: Container; statusLine: StatusLineComponent; // Session access @@ -226,7 +233,7 @@ export interface InteractiveModeContext { reason: "start" | "switch" | "branch" | "tree" | "shutdown", previousSessionFile?: string, ): Promise; - setHookWidget(key: string, content: unknown): void; + setHookWidget(key: string, content: ExtensionWidgetContent, options?: ExtensionWidgetOptions): void; setHookStatus(key: string, text: string | undefined): void; showHookSelector( title: string, diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 8bd67ddfe..efdf4fa0c 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -455,18 +455,6 @@ function maybeWarnSuspiciousUnicodeEscapePlaceholder(edits: HashlineEdit[], warn // Edit Application // ═══════════════════════════════════════════════════════════════════════════ -const MIN_AUTOCORRECT_LENGTH = 2; - -function shouldAutocorrect(line: string, otherLine: string): boolean { - if (!line || line !== otherLine) return false; - line = line.trim(); - if (line.length < MIN_AUTOCORRECT_LENGTH) { - // if brace, we allow - return line.endsWith("}") || line.endsWith(")"); - } - return true; -} - /** * Apply an array of hashline edits to file content. * @@ -635,34 +623,7 @@ export function applyHashlineEdits( trackFirstChanged(edit.pos.line); } else { const count = edit.end.line - edit.pos.line + 1; - const newLines = [...edit.lines]; - const trailingReplacementLine = newLines[newLines.length - 1]?.trimEnd(); - const nextSurvivingLine = fileLines[edit.end.line]?.trimEnd(); - if ( - shouldAutocorrect(trailingReplacementLine, nextSurvivingLine) && - // Safety: only correct when end-line content differs from the duplicate. - // If end already points to the boundary, matching next line is coincidence. - fileLines[edit.end.line - 1]?.trimEnd() !== trailingReplacementLine - ) { - newLines.pop(); - warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed trailing replacement line "${trailingReplacementLine}" that duplicated next surviving line`, - ); - } - const leadingReplacementLine = newLines[0]?.trimEnd(); - const prevSurvivingLine = fileLines[edit.pos.line - 2]?.trimEnd(); - if ( - shouldAutocorrect(leadingReplacementLine, prevSurvivingLine) && - // Safety: only correct when pos-line content differs from the duplicate. - // If pos already points to the boundary, matching prev line is coincidence. - fileLines[edit.pos.line - 1]?.trimEnd() !== leadingReplacementLine - ) { - newLines.shift(); - warnings.push( - `Auto-corrected range replace ${edit.pos.line}#${edit.pos.hash}-${edit.end.line}#${edit.end.hash}: removed leading replacement line "${leadingReplacementLine}" that duplicated preceding surviving line`, - ); - } - fileLines.splice(edit.pos.line - 1, count, ...newLines); + fileLines.splice(edit.pos.line - 1, count, ...edit.lines); trackFirstChanged(edit.pos.line); } break; diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 6f41ad153..c2413c737 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -13,6 +13,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { $env, getAgentDbPath, getAgentDir, getProjectDir, logger, postmortem } from "@oh-my-pi/pi-utils"; import chalk from "chalk"; import { AsyncJobManager } from "./async"; +import { createAutoresearchExtension } from "./autoresearch"; import { loadCapability } from "./capability"; import { type Rule, ruleCapability } from "./capability/rule"; import { ModelRegistry } from "./config/model-registry"; @@ -1010,6 +1011,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } const inlineExtensions: ExtensionFactory[] = options.extensions ? [...options.extensions] : []; + inlineExtensions.push(createAutoresearchExtension); if (customTools.length > 0) { inlineExtensions.push(createCustomToolsExtension(customTools)); } diff --git a/packages/coding-agent/test/autoresearch-state.test.ts b/packages/coding-agent/test/autoresearch-state.test.ts new file mode 100644 index 000000000..0f50e4c66 --- /dev/null +++ b/packages/coding-agent/test/autoresearch-state.test.ts @@ -0,0 +1,107 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { isAutoresearchShCommand } from "../src/autoresearch/helpers"; +import { reconstructStateFromJsonl } from "../src/autoresearch/state"; + +function makeTempDir(): string { + const dir = path.join(os.tmpdir(), `pi-autoresearch-test-${Snowflake.next()}`); + fs.mkdirSync(dir, { recursive: true }); + return dir; +} + +describe("autoresearch state reconstruction", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("reconstructs the latest segment and current metric definitions from autoresearch.jsonl", () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const jsonlPath = path.join(dir, "autoresearch.jsonl"); + fs.writeFileSync( + jsonlPath, + [ + JSON.stringify({ + type: "config", + name: "First", + metricName: "runtime_ms", + metricUnit: "ms", + bestDirection: "lower", + }), + JSON.stringify({ + commit: "aaaaaaa", + metric: 100, + metrics: { memory_mb: 32 }, + status: "keep", + description: "baseline", + timestamp: 1, + }), + JSON.stringify({ + commit: "bbbbbbb", + metric: 90, + metrics: { memory_mb: 30 }, + status: "keep", + description: "improved", + timestamp: 2, + }), + JSON.stringify({ + type: "config", + name: "Second", + metricName: "throughput", + metricUnit: "", + bestDirection: "higher", + }), + JSON.stringify({ + commit: "ccccccc", + metric: 1200, + metrics: { latency_ms: 15 }, + status: "keep", + description: "new baseline", + timestamp: 3, + }), + JSON.stringify({ + commit: "ddddddd", + metric: 1320, + metrics: { latency_ms: 18 }, + status: "discard", + description: "regressed latency", + timestamp: 4, + }), + ].join("\n"), + ); + + const reconstructed = reconstructStateFromJsonl(dir); + const state = reconstructed.state; + + expect(reconstructed.hasLog).toBe(true); + expect(state.name).toBe("Second"); + expect(state.metricName).toBe("throughput"); + expect(state.bestDirection).toBe("higher"); + expect(state.currentSegment).toBe(1); + expect(state.bestMetric).toBe(1200); + expect(state.results).toHaveLength(4); + expect(state.results.filter(result => result.segment === 1)).toHaveLength(2); + expect(state.secondaryMetrics).toEqual([{ name: "latency_ms", unit: "ms" }]); + }); +}); + +describe("autoresearch command guard", () => { + it("accepts autoresearch.sh through common wrappers", () => { + expect(isAutoresearchShCommand("bash autoresearch.sh")).toBe(true); + expect(isAutoresearchShCommand("FOO=bar time bash ./autoresearch.sh --quick")).toBe(true); + expect(isAutoresearchShCommand("nice -n 10 /tmp/project/autoresearch.sh")).toBe(true); + }); + + it("rejects commands where autoresearch.sh is not the first real command", () => { + expect(isAutoresearchShCommand("python script.py && ./autoresearch.sh")).toBe(false); + expect(isAutoresearchShCommand("echo hi; autoresearch.sh")).toBe(false); + expect(isAutoresearchShCommand("bash -lc 'autoresearch.sh'")).toBe(false); + }); +}); diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 0e6794de4..9c961c13b 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -519,7 +519,7 @@ describe("applyHashlineEdits — heuristics", () => { expect(result.lines).toBe("aaa\nBBB\nccc"); }); - it("auto-corrects off-by-one range end that duplicates a closing brace", () => { + it("preserves duplicated trailing closer lines exactly as provided", () => { const content = "if (ok) {\n run();\n}\nafter();"; const edits: HashlineEdit[] = [ { @@ -530,29 +530,11 @@ describe("applyHashlineEdits — heuristics", () => { }, ]; const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("if (ok) {\n runSafe();\n}\nafter();"); - expect(result.warnings).toHaveLength(1); - expect(result.warnings?.[0]).toContain("Auto-corrected range replace"); - expect(result.warnings?.[0]).toContain('"}"'); + expect(result.lines).toBe("if (ok) {\n runSafe();\n}\n}\nafter();"); + expect(result.warnings).toBeUndefined(); }); - it('auto-corrects off-by-one range end that duplicates a ");" closer', () => { - const content = "doThing(\n value,\n);\nnext();"; - const edits: HashlineEdit[] = [ - { - op: "replace", - pos: makeTag(1, "doThing("), - end: makeTag(2, " value,"), - lines: ["doThing(", " normalize(value),", ");"], - }, - ]; - const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("doThing(\n normalize(value),\n);\nnext();"); - expect(result.warnings).toHaveLength(1); - expect(result.warnings?.[0]).toContain('");"'); - }); - - it("auto-corrects duplicated trailing lines when they match the next surviving line", () => { + it("preserves duplicated trailing content when replacement re-emits the next line", () => { const content = "start\n oldCall();\nnextCall();\nafter();"; const edits: HashlineEdit[] = [ { @@ -563,12 +545,11 @@ describe("applyHashlineEdits — heuristics", () => { }, ]; const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("start\n newCall();\nnextCall();\nafter();"); - expect(result.warnings).toHaveLength(1); - expect(result.warnings?.[0]).toContain("removed trailing replacement line"); + expect(result.lines).toBe("start\n newCall();\nnextCall();\nnextCall();\nafter();"); + expect(result.warnings).toBeUndefined(); }); - it("auto-corrects off-by-one range start that duplicates a preceding line", () => { + it("preserves duplicated leading content when replacement re-emits the previous line", () => { const content = "if (x) {\n oldBody();\n}\nafter();"; const edits: HashlineEdit[] = [ { @@ -579,10 +560,10 @@ describe("applyHashlineEdits — heuristics", () => { }, ]; const result = applyHashlineEdits(content, edits); - expect(result.lines).toBe("if (x) {\n newBody();\n}\nafter();"); - expect(result.warnings).toHaveLength(1); - expect(result.warnings?.[0]).toContain("removed leading replacement line"); + expect(result.lines).toBe("if (x) {\nif (x) {\n newBody();\n}\nafter();"); + expect(result.warnings).toBeUndefined(); }); + it("auto-corrects leading escaped tab indentation by default", () => { const previous = Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS; delete Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS; @@ -614,7 +595,7 @@ describe("applyHashlineEdits — heuristics", () => { } }); - it("does not auto-correct when edit already includes real tab characters", () => { + it("preserves mixed real-tab and escaped-tab content verbatim", () => { const previous = Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS; delete Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS; try { @@ -634,6 +615,7 @@ describe("applyHashlineEdits — heuristics", () => { else Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS = previous; } }); + it("warns on literal \\uDDDD without changing content", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), lines: ["\\uDDDD"] }]; From cf5b62a5abc33c07f6cc64164fa9ad8e0aa3b4d3 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 17:03:53 +0100 Subject: [PATCH 06/22] chore: configured Biome linter rules for import and type checking - Enabled noUnusedImports rule and disabled noVoidTypeReturn in Biome linter configuration. --- biome.json | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/biome.json b/biome.json index 47c5d9045..b7b152eb1 100644 --- a/biome.json +++ b/biome.json @@ -3,6 +3,10 @@ "enabled": true, "rules": { "recommended": true, + "correctness": { + "noUnusedImports": "error", + "noVoidTypeReturn": "off" + }, "style": { "noNonNullAssertion": "off", "useConst": "error", From c9a7bb0c969501d1caf5fa6608e93b1b80bff125 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 17:17:13 +0100 Subject: [PATCH 07/22] feat(patch): restructured hashline edits with explicit replace_line/replace_range and boundary operations - Refactored hashline edit operations from implicit `replace` to explicit `replace_line` and `replace_range` with mandatory `end` parameter for ranges. - Added file-level operations `append_eof` and `prepend_bof` for boundary insertions, making `pos` required for anchor-based operations. - Enforced stricter anchor validation per operation type and simplified edit application logic by separating file-level from anchor-based operations. - Updated hashline edit application to preserve duplicated boundary lines without auto-correction, changing previous behavior. --- packages/coding-agent/CHANGELOG.md | 7 + packages/coding-agent/src/patch/hashline.ts | 187 ++++++++++-------- packages/coding-agent/src/patch/index.ts | 81 +++++--- .../src/prompts/tools/hashline.md | 38 ++-- 4 files changed, 181 insertions(+), 132 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index fde575753..68513549f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,11 @@ # Changelog ## [Unreleased] +### Breaking Changes + +- Changed hashline edit operation types from `replace` (with optional `end`) to explicit `replace_line` and `replace_range` operations +- Added required `append_eof` and `prepend_bof` operations for file-level edits; `append` and `prepend` now require an anchor position +- Made `pos` parameter required for `replace_line`, `append`, and `prepend` operations; `append_eof` and `prepend_bof` no longer accept anchors ### Added @@ -25,6 +30,8 @@ ### Changed +- Refactored hashline edit validation to enforce stricter anchor requirements per operation type +- Updated edit application logic to handle explicit file-level operations (`append_eof`, `prepend_bof`) separately from anchor-based operations - Changed `setWidget` API to accept `ExtensionWidgetOptions` parameter for placement control - Changed widget placement logic to manage widgets above and below editor separately - Changed hashline edit application to preserve duplicated boundary lines exactly as provided instead of auto-correcting them diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index efdf4fa0c..89e1df7e7 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -16,9 +16,12 @@ import type { HashMismatch } from "./types"; export type Anchor = { line: number; hash: string }; export type HashlineEdit = - | { op: "replace"; pos: Anchor; end?: Anchor; lines: string[] } - | { op: "append"; pos?: Anchor; lines: string[] } - | { op: "prepend"; pos?: Anchor; lines: string[] }; + | { op: "replace_line"; pos: Anchor; lines: string[] } + | { op: "replace_range"; pos: Anchor; end: Anchor; lines: string[] } + | { op: "append"; pos: Anchor; lines: string[] } + | { op: "prepend"; pos: Anchor; lines: string[] } + | { op: "append_eof"; lines: string[] } + | { op: "prepend_bof"; lines: string[] }; const NIBBLE_STR = "ZPMQVRWSNKTXJBYH"; @@ -501,28 +504,29 @@ export function applyHashlineEdits( } for (const edit of edits) { switch (edit.op) { - case "replace": { - if (edit.end) { - const startValid = validateRef(edit.pos); - const endValid = validateRef(edit.end); - if (!startValid || !endValid) continue; - if (edit.pos.line > edit.end.line) { - throw new Error(`Range start line ${edit.pos.line} must be <= end line ${edit.end.line}`); - } - } else { - if (!validateRef(edit.pos)) continue; + case "replace_line": { + if (!validateRef(edit.pos)) continue; + break; + } + case "replace_range": { + const startValid = validateRef(edit.pos); + const endValid = validateRef(edit.end); + if (!startValid || !endValid) continue; + if (edit.pos.line > edit.end.line) { + throw new Error(`Range start line ${edit.pos.line} must be <= end line ${edit.end.line}`); } break; } - case "append": { - if (edit.pos && !validateRef(edit.pos)) continue; + case "append": + case "prepend": { + if (!validateRef(edit.pos)) continue; if (edit.lines.length === 0) { edit.lines = [""]; // insert an empty line } break; } - case "prepend": { - if (edit.pos && !validateRef(edit.pos)) continue; + case "append_eof": + case "prepend_bof": { if (edit.lines.length === 0) { edit.lines = [""]; // insert an empty line } @@ -542,25 +546,22 @@ export function applyHashlineEdits( const edit = edits[i]; let lineKey: string; switch (edit.op) { - case "replace": - if (!edit.end) { - lineKey = `s:${edit.pos.line}`; - } else { - lineKey = `r:${edit.pos.line}:${edit.end.line}`; - } + case "replace_line": + lineKey = `s:${edit.pos.line}`; + break; + case "replace_range": + lineKey = `r:${edit.pos.line}:${edit.end.line}`; break; case "append": - if (edit.pos) { - lineKey = `i:${edit.pos.line}`; - break; - } - lineKey = "ieof"; + lineKey = `i:${edit.pos.line}`; break; case "prepend": - if (edit.pos) { - lineKey = `ib:${edit.pos.line}`; - break; - } + lineKey = `ib:${edit.pos.line}`; + break; + case "append_eof": + lineKey = "ieof"; + break; + case "prepend_bof": lineKey = "ibef"; break; } @@ -582,20 +583,28 @@ export function applyHashlineEdits( let sortLine: number; let precedence: number; switch (edit.op) { - case "replace": - if (!edit.end) { - sortLine = edit.pos.line; - } else { - sortLine = edit.end.line; - } + case "replace_line": + sortLine = edit.pos.line; + precedence = 0; + break; + case "replace_range": + sortLine = edit.end.line; precedence = 0; break; case "append": - sortLine = edit.pos ? edit.pos.line : fileLines.length + 1; + sortLine = edit.pos.line; precedence = 1; break; case "prepend": - sortLine = edit.pos ? edit.pos.line : 0; + sortLine = edit.pos.line; + precedence = 2; + break; + case "append_eof": + sortLine = fileLines.length + 1; + precedence = 1; + break; + case "prepend_bof": + sortLine = 0; precedence = 2; break; } @@ -607,25 +616,25 @@ export function applyHashlineEdits( // Apply edits bottom-up for (const { edit, idx } of annotated) { switch (edit.op) { - case "replace": { - if (!edit.end) { - const origLines = originalFileLines.slice(edit.pos.line - 1, edit.pos.line); - const newLines = edit.lines; - if (origLines.length === newLines.length && origLines.every((line, i) => line === newLines[i])) { - noopEdits.push({ - editIndex: idx, - loc: `${edit.pos.line}#${edit.pos.hash}`, - current: origLines.join("\n"), - }); - break; - } - fileLines.splice(edit.pos.line - 1, 1, ...newLines); - trackFirstChanged(edit.pos.line); - } else { - const count = edit.end.line - edit.pos.line + 1; - fileLines.splice(edit.pos.line - 1, count, ...edit.lines); - trackFirstChanged(edit.pos.line); + case "replace_line": { + const origLines = originalFileLines.slice(edit.pos.line - 1, edit.pos.line); + const newLines = edit.lines; + if (origLines.length === newLines.length && origLines.every((line, i) => line === newLines[i])) { + noopEdits.push({ + editIndex: idx, + loc: `${edit.pos.line}#${edit.pos.hash}`, + current: origLines.join("\n"), + }); + break; } + fileLines.splice(edit.pos.line - 1, 1, ...newLines); + trackFirstChanged(edit.pos.line); + break; + } + case "replace_range": { + const count = edit.end.line - edit.pos.line + 1; + fileLines.splice(edit.pos.line - 1, count, ...edit.lines); + trackFirstChanged(edit.pos.line); break; } case "append": { @@ -633,23 +642,13 @@ export function applyHashlineEdits( if (inserted.length === 0) { noopEdits.push({ editIndex: idx, - loc: edit.pos ? `${edit.pos.line}#${edit.pos.hash}` : "EOF", - current: edit.pos ? originalFileLines[edit.pos.line - 1] : "", + loc: `${edit.pos.line}#${edit.pos.hash}`, + current: originalFileLines[edit.pos.line - 1], }); break; } - if (edit.pos) { - fileLines.splice(edit.pos.line, 0, ...inserted); - trackFirstChanged(edit.pos.line + 1); - } else { - if (fileLines.length === 1 && fileLines[0] === "") { - fileLines.splice(0, 1, ...inserted); - trackFirstChanged(1); - } else { - fileLines.splice(fileLines.length, 0, ...inserted); - trackFirstChanged(fileLines.length - inserted.length + 1); - } - } + fileLines.splice(edit.pos.line, 0, ...inserted); + trackFirstChanged(edit.pos.line + 1); break; } case "prepend": { @@ -657,22 +656,42 @@ export function applyHashlineEdits( if (inserted.length === 0) { noopEdits.push({ editIndex: idx, - loc: edit.pos ? `${edit.pos.line}#${edit.pos.hash}` : "BOF", - current: edit.pos ? originalFileLines[edit.pos.line - 1] : "", + loc: `${edit.pos.line}#${edit.pos.hash}`, + current: originalFileLines[edit.pos.line - 1], }); break; } - if (edit.pos) { - fileLines.splice(edit.pos.line - 1, 0, ...inserted); - trackFirstChanged(edit.pos.line); - } else { - if (fileLines.length === 1 && fileLines[0] === "") { - fileLines.splice(0, 1, ...inserted); - } else { - fileLines.splice(0, 0, ...inserted); - } - trackFirstChanged(1); + fileLines.splice(edit.pos.line - 1, 0, ...inserted); + trackFirstChanged(edit.pos.line); + break; + } + case "append_eof": { + const inserted = edit.lines; + if (inserted.length === 0) { + noopEdits.push({ editIndex: idx, loc: "EOF", current: "" }); + break; } + if (fileLines.length === 1 && fileLines[0] === "") { + fileLines.splice(0, 1, ...inserted); + trackFirstChanged(1); + } else { + fileLines.splice(fileLines.length, 0, ...inserted); + trackFirstChanged(fileLines.length - inserted.length + 1); + } + break; + } + case "prepend_bof": { + const inserted = edit.lines; + if (inserted.length === 0) { + noopEdits.push({ editIndex: idx, loc: "BOF", current: "" }); + break; + } + if (fileLines.length === 1 && fileLines[0] === "") { + fileLines.splice(0, 1, ...inserted); + } else { + fileLines.splice(0, 0, ...inserted); + } + trackFirstChanged(1); break; } } diff --git a/packages/coding-agent/src/patch/index.ts b/packages/coding-agent/src/patch/index.ts index 4576ab7e5..38651e2a8 100644 --- a/packages/coding-agent/src/patch/index.ts +++ b/packages/coding-agent/src/patch/index.ts @@ -176,7 +176,7 @@ export function hashlineParseText(edit: string[] | string | null): string[] { const hashlineEditSchema = Type.Object( { - op: StringEnum(["replace", "append", "prepend"]), + op: StringEnum(["replace_line", "replace_range", "append", "prepend", "append_eof", "prepend_bof"]), pos: Type.Optional(Type.String({ description: "anchor" })), end: Type.Optional(Type.String({ description: "limit position" })), lines: Type.Union([ @@ -209,13 +209,14 @@ export type HashlineParams = Static; * Map flat tool-schema edits (tag/end) into typed HashlineEdit objects. * * Resilient: as long as at least one anchor exists, we execute. - * - replace + tag only → single-line replace - * - replace + tag + end → range replace + * - replace_line + tag → single-line replace + * - replace_range + tag + end → range replace * - append + tag or end → append after that anchor * - prepend + tag or end → prepend before that anchor - * - no anchors → file-level append/prepend (only for those ops) + * - append_eof → file-level append (no anchors needed) + * - prepend_bof → file-level prepend (no anchors needed) * - * Unknown ops default to "replace". + * Unknown ops default to replace_line/replace_range based on available anchors. */ function resolveEditAnchors(edits: HashlineToolEdit[]): HashlineEdit[] { const result: HashlineEdit[] = []; @@ -224,25 +225,47 @@ function resolveEditAnchors(edits: HashlineToolEdit[]): HashlineEdit[] { const tag = edit.pos ? tryParseTag(edit.pos) : undefined; const end = edit.end ? tryParseTag(edit.end) : undefined; - // Normalize op — default unknown values to "replace" - const op = edit.op === "append" || edit.op === "prepend" ? edit.op : "replace"; - switch (op) { - case "replace": { - if (tag && end) { - result.push({ op: "replace", pos: tag, end, lines }); - } else if (tag || end) { - result.push({ op: "replace", pos: tag || end!, lines }); - } else { - throw new Error("Replace requires at least one anchor (tag or end)."); - } + switch (edit.op) { + case "replace_line": { + const anchor = tag ?? end; + if (!anchor) throw new Error("replace_line requires an anchor (pos)."); + result.push({ op: "replace_line", pos: anchor, lines }); + break; + } + case "replace_range": { + if (!tag || !end) throw new Error("replace_range requires both pos and end anchors."); + result.push({ op: "replace_range", pos: tag, end, lines }); break; } case "append": { - result.push({ op: "append", pos: tag ?? end, lines }); + const anchor = tag ?? end; + if (!anchor) throw new Error("append requires an anchor (pos)."); + result.push({ op: "append", pos: anchor, lines }); break; } case "prepend": { - result.push({ op: "prepend", pos: end ?? tag, lines }); + const anchor = end ?? tag; + if (!anchor) throw new Error("prepend requires an anchor (pos)."); + result.push({ op: "prepend", pos: anchor, lines }); + break; + } + case "append_eof": { + result.push({ op: "append_eof", lines }); + break; + } + case "prepend_bof": { + result.push({ op: "prepend_bof", lines }); + break; + } + default: { + // Backward compat for stale model output: infer op from available anchors + if (tag && end) { + result.push({ op: "replace_range", pos: tag, end, lines }); + } else if (tag || end) { + result.push({ op: "replace_line", pos: (tag ?? end)!, lines }); + } else { + throw new Error("Unknown op requires at least one anchor (pos or end)."); + } break; } } @@ -552,8 +575,8 @@ export class EditTool implements AgentTool { const lines: string[] = []; for (const edit of edits) { // For file creation, only anchorless appends/prepends are valid - if ((edit.op === "append" || edit.op === "prepend") && !edit.pos && !edit.end) { - if (edit.op === "prepend") { + if (edit.op === "append_eof" || edit.op === "prepend_bof") { + if (edit.op === "prepend_bof") { lines.unshift(...hashlineParseText(edit.lines)); } else { lines.push(...hashlineParseText(edit.lines)); @@ -612,20 +635,18 @@ export class EditTool implements AgentTool { for (const edit of anchorEdits) { refs.length = 0; switch (edit.op) { - case "replace": - if (edit.end) { - refs.push(edit.end, edit.pos); - } else { - refs.push(edit.pos); - } + case "replace_line": + refs.push(edit.pos); + break; + case "replace_range": + refs.push(edit.end, edit.pos); break; case "append": - if (edit.pos) refs.push(edit.pos); - break; case "prepend": - if (edit.pos) refs.push(edit.pos); + refs.push(edit.pos); break; - default: + case "append_eof": + case "prepend_bof": break; } diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index 2b7fea2e3..ce3342160 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -8,15 +8,17 @@ Read the file first to get fresh tags. Submit one `edit` call per file with all **`delete`** — if true, delete the file. **`edits[n].pos`** — the anchor line. Meaning depends on `op`: - - if `replace`: first line to rewrite - - if `prepend`: line to insert new lines **before**; omit for beginning of file - - if `append`: line to insert new lines **after**; omit for end of file -**`edits[n].end`** — range replace only. The last line of the range (inclusive). Omit for single-line replace. + - if `replace_line`: the line to rewrite + - if `replace_range`: first line of the range to rewrite + - if `prepend`: line to insert new lines **before** + - if `append`: line to insert new lines **after** + - Not used by `append_eof` or `prepend_bof`. +**`edits[n].end`** — only used by `replace_range`: the last line of the range (inclusive). **`edits[n].lines`** — the replacement content: - - for `replace`: the exact lines that will replace `[pos, end??pos]` inclusively (or the single `pos` line when `end` is omitted) - - for `prepend`/`append`: the new lines to insert + - for `replace_line`/`replace_range`: the exact lines that will replace the target line(s) + - for `append`/`prepend`/`append_eof`/`prepend_bof`: the new lines to insert - `[""]` — blank line - - `null` or `[]` — delete if replace + - `null` or `[]` — delete if `replace_line`/`replace_range` - If `lines` contains content that already exists after `end`, those lines **will be duplicated** in the output. - Keep `lines` to exactly what belongs inside the consumed range. - Ops are applied bottom-up. Tags **MUST** be referenced from the most recent `read` output. @@ -51,7 +53,7 @@ Change the timeout from `5000` to `30_000`: { path: "util.ts", edits: [{ - op: "replace", + op: "replace_line", pos: {{hlineref 2 "const timeout = 5000;"}}, lines: ["const timeout = 30_000;"] }] @@ -65,7 +67,7 @@ Single line — `lines: null` deletes entirely: { path: "util.ts", edits: [{ - op: "replace", + op: "replace_line", pos: {{hlineref 1 "// @ts-ignore"}}, lines: null }] @@ -76,7 +78,7 @@ Range — remove the legacy block (lines 10–11): { path: "util.ts", edits: [{ - op: "replace", + op: "replace_range", pos: {{hlineref 10 "\t// TODO: remove after migration"}}, end: {{hlineref 11 "\tlegacy();"}}, lines: null @@ -93,7 +95,7 @@ When changing body content, replace the **entire** body span — not just one li { path: "util.ts", edits: [{ - op: "replace", + op: "replace_range", pos: {{hlineref 15 "\t\tconsole.error(err);"}}, end: {{hlineref 16 "\t\treturn null;"}}, lines: [ @@ -113,7 +115,7 @@ Bad — `end` stops at the inner `\t}` on line 17, so the outer `}` on line 18 s { path: "util.ts", edits: [{ - op: "replace", + op: "replace_range", pos: {{hlineref 9 "function beta() {"}}, end: {{hlineref 17 "\t}"}}, lines: [ @@ -129,7 +131,7 @@ Good — `end` includes the function's own `}` on line 18, so the old closer is { path: "util.ts", edits: [{ - op: "replace", + op: "replace_range", pos: {{hlineref 9 "function beta() {"}}, end: {{hlineref 18 "}"}}, lines: [ @@ -143,7 +145,7 @@ Good — `end` includes the function's own `}` on line 18, so the old closer is -Do not anchor `replace` on a mixed boundary line such as `} catch (err) {`, `} else {`, `}),`, or `},{`. Those lines belong to two adjacent structures at once. +Do not anchor `replace_range` on a mixed boundary line such as `} catch (err) {`, `} else {`, `}),`, or `},{`. Those lines belong to two adjacent structures at once. Bad — if you need to change code on both sides of that line, replacing just the boundary span will usually leave one side's syntax behind. @@ -176,12 +178,12 @@ Use a trailing `""` to preserve the blank line between sibling declarations. - You **MUST NOT** use this tool to reformat, reindent, or adjust whitespace — run the project's formatter instead. - Every tag **MUST** be copied exactly from your most recent `read` output as `N#ID`. Stale or mistyped tags cause mismatches. -- Edit payload: `{ path, edits[] }`. Each entry: `op`, `lines`, optional `pos`/`end`. No extra keys. -- For `append`/`prepend`, `lines` **MUST** contain only the newly introduced content. Do not re-emit surrounding content, or terminators that already exist. -- When changing existing code near a block tail or closing delimiter, default to `replace` over the owned span instead of inserting around the boundary. +- Edit payload: `{ path, edits[] }`. Each entry: `op`, `lines`, plus `pos` and/or `end` depending on op. `replace_line`/`append`/`prepend` require `pos`. `replace_range` requires both `pos` and `end`. `append_eof`/`prepend_bof` require neither. No extra keys. +- For `append`/`prepend`/`append_eof`/`prepend_bof`, `lines` **MUST** contain only the newly introduced content. Do not re-emit surrounding content, or terminators that already exist. +- When changing existing code near a block tail or closing delimiter, default to `replace_range` over the owned span instead of inserting around the boundary. - When adding a sibling declaration, default to `prepend` on the next sibling declaration instead of `append` on the previous block's closing brace. - **Block boundaries travel together.** For a block `{ header / body / closer }`, there are exactly two valid replace shapes: (a) replace only the body — `pos`=first body line, `end`=last body line, leave the header and closer untouched; or (b) replace the whole block — `pos`=header, `end`=closer, re-emit all three in `lines`. Never split them: do not set `end` to the closer while omitting it from `lines` (deletes it), and do not emit the closer in `lines` without including it in `end` (duplicates it). This applies to every block terminator: `}`, `continue`, `break`, `return`, `throw`. -- **Never target shared boundary lines.** Do not use `replace` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct including its true trailing delimiter. +- **Never target shared boundary lines.** Do not use `replace_range` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct including its true trailing delimiter. - **`lines` must not extend past `end`.** `lines` replaces exactly `pos..end`. Content after `end` survives. If you include lines in `lines` that exist after `end`, they will appear twice. Either extend `end` to cover all lines you are re-emitting, or remove the extra lines from `lines`. - `lines` entries **MUST** be literal file content with indentation copied exactly from the `read` output. If the file uses tabs, use a real tab character. - After any successful `edit` call on a file, the next change to that same file **MUST** start with a fresh `read`. Do not chain a second `edit` call off stale mental state, even if the intended range is nearby. From 7fb18faf4c65d6f01ed478be6a67ad947a09db8d Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 17:57:21 +0100 Subject: [PATCH 08/22] fix(coding-agent): backported pi-mono changes (1feccfed..b21b42d0) packages/ai: - feat: expose provider responseId on AssistantMessage - feat: lazy-load provider modules for faster startup - fix: hash foreign Responses API tool call IDs exceeding 64-char limit - fix: ignore null chunks in openai-completions streams - fix: keep image tool results inline for Gemini 3+ and Antigravity - fix: correct Bedrock Claude 4.6 context window to 200k - fix: support prompt caching for Bedrock application inference profiles - fix: add OpenRouter reasoning payload format - fix: ignore placeholder Vertex API keys - fix: skip AJV validation in restricted runtimes - fix: Anthropic OAuth client injection and responseId extraction - fix: Codex incomplete/failed response status handling packages/agent: - fix: defer steering until after tool execution completes packages/tui: - feat: namespaced keybinding IDs with KeybindingsManager conflict detection - feat: configurable select list column sizing (#2154 by @markusylisiurunen) - fix: stream truncateToWidth for large strings - fix: skip Termux height redraws - fix: stop evicting unrelated default keybindings - fix: resolve raw backspace ambiguity on Windows Terminal - fix: clear stale scrollback on session switch (#2155 by @Perlence) - fix: remove trailing markdown block spacing (#2152 by @markusylisiurunen) packages/coding-agent: - feat: add resizable share sidebar (#2435 by @dmmulroy) - feat: emit OSC 133 command-executed marker - feat: reload custom themes from disk watcher - feat: add --fork session flag - feat: file mutation queue for serialized writes - feat: initial message consolidation utility - fix: keybindings migrated to namespaced IDs - fix: resolve waitForRetry() race when auto-retry produces tool calls - fix: handle slash-delimited /model refs - fix: refresh active model after provider updates - fix: extended transient error patterns for retry --- docs/porting-from-pi-mono.md | 6 +- packages/agent/src/agent-loop.ts | 64 +- packages/agent/test/agent-loop.test.ts | 55 +- packages/ai/src/index.ts | 1 + packages/ai/src/providers/amazon-bedrock.ts | 10 + packages/ai/src/providers/anthropic.ts | 43 +- packages/ai/src/providers/google-shared.ts | 29 +- .../src/providers/openai-codex-responses.ts | 6 +- .../providers/openai-completions-compat.ts | 8 +- .../ai/src/providers/openai-completions.ts | 49 +- .../src/providers/openai-responses-shared.ts | 27 +- .../ai/src/providers/register-builtins.ts | 309 +++++++++ packages/ai/src/types.ts | 5 +- packages/coding-agent/src/cli/args.ts | 3 + .../coding-agent/src/cli/initial-message.ts | 58 ++ .../coding-agent/src/config/keybindings.ts | 627 ++++++++++++------ .../coding-agent/src/config/model-registry.ts | 1 + .../coding-agent/src/config/model-resolver.ts | 66 +- .../coding-agent/src/export/html/template.css | 56 +- .../src/export/html/template.generated.ts | 2 +- .../src/export/html/template.html | 1 + .../coding-agent/src/export/html/template.js | 107 +++ .../src/extensibility/extensions/types.ts | 2 +- packages/coding-agent/src/main.ts | 78 +-- .../src/modes/components/custom-editor.ts | 94 +-- .../src/modes/components/keybinding-hints.ts | 18 +- .../src/modes/components/login-dialog.ts | 6 +- .../src/modes/components/user-message.ts | 16 + .../src/modes/controllers/input-controller.ts | 52 +- .../src/modes/interactive-mode.ts | 2 +- packages/coding-agent/src/modes/print-mode.ts | 2 +- .../src/modes/prompt-action-autocomplete.ts | 14 +- .../coding-agent/src/modes/theme/theme.ts | 99 +-- .../src/modes/utils/hotkeys-markdown.ts | 38 +- .../src/modes/utils/ui-helpers.ts | 2 +- .../coding-agent/src/session/agent-session.ts | 6 +- .../coding-agent/src/utils/child-process.ts | 88 +++ packages/coding-agent/test/args.test.ts | 7 + .../coding-agent/test/core/hashline.test.ts | 102 +-- .../coding-agent/test/initial-message.test.ts | 40 ++ .../test/keybindings-display.test.ts | 12 +- .../test/keybindings-migration.test.ts | 50 ++ .../command-controller-hotkeys.test.ts | 34 +- .../test/prompt-action-autocomplete.test.ts | 26 +- .../tui/src/components/cancellable-loader.ts | 5 +- packages/tui/src/components/editor.ts | 115 ++-- packages/tui/src/components/input.ts | 38 +- packages/tui/src/components/markdown.ts | 27 +- packages/tui/src/components/select-list.ts | 194 +++--- packages/tui/src/components/settings-list.ts | 11 +- packages/tui/src/keybindings.ts | 350 ++++++---- packages/tui/src/keys.ts | 28 + packages/tui/src/tui.ts | 27 +- packages/tui/src/utils.ts | 12 +- packages/tui/test/editor.test.ts | 10 +- packages/tui/test/keybindings.test.ts | 37 ++ packages/tui/test/markdown.test.ts | 78 +++ packages/tui/test/select-list.test.ts | 171 +++++ packages/tui/test/settings-list.test.ts | 47 ++ packages/tui/test/truncate-to-width.test.ts | 49 ++ 60 files changed, 2581 insertions(+), 939 deletions(-) create mode 100644 packages/ai/src/providers/register-builtins.ts create mode 100644 packages/coding-agent/src/cli/initial-message.ts create mode 100644 packages/coding-agent/src/utils/child-process.ts create mode 100644 packages/coding-agent/test/initial-message.test.ts create mode 100644 packages/coding-agent/test/keybindings-migration.test.ts create mode 100644 packages/tui/test/keybindings.test.ts create mode 100644 packages/tui/test/select-list.test.ts create mode 100644 packages/tui/test/settings-list.test.ts create mode 100644 packages/tui/test/truncate-to-width.test.ts diff --git a/docs/porting-from-pi-mono.md b/docs/porting-from-pi-mono.md index fbc41b604..24a9f8704 100644 --- a/docs/porting-from-pi-mono.md +++ b/docs/porting-from-pi-mono.md @@ -5,15 +5,15 @@ Use it for any merge: single file, feature branch, or full release sync. ## Last Sync Point -**Commit:** `1feccfedcb1eeeca91be0b9d389e8e5a9daee505` -**Date:** 2026-03-14 +**Commit:** `b21b42d032919de2f2e6920a76fa9a37c3920c0a` +**Date:** 2026-03-22 Update this section after each sync; do not reuse the previous range. When starting a new sync, generate patches from this commit forward: ```bash -git format-patch 15e0957b045d9e0d49253b2285cb585cf3a75c55..HEAD --stdout > changes.patch +git format-patch b21b42d032919de2f2e6920a76fa9a37c3920c0a..HEAD --stdout > changes.patch ``` ## 0) Define the scope diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 0eac54e36..e32feaea3 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -196,7 +196,6 @@ async function runLoop( // Outer loop: continues when queued follow-up messages arrive after agent would stop while (true) { let hasMoreToolCalls = true; - let steeringAfterTools: AgentMessage[] | null = null; // Inner loop: process tool calls and steering messages while (hasMoreToolCalls || pendingMessages.length > 0) { @@ -225,6 +224,7 @@ async function runLoop( // Stream assistant response const message = await streamAssistantResponse(currentContext, config, signal, stream, streamFn); newMessages.push(message); + let steeringMessagesFromExecution: AgentMessage[] | undefined; if (message.stopReason === "error" || message.stopReason === "aborted") { // Create placeholder tool results for any tool calls in the aborted message @@ -250,19 +250,20 @@ async function runLoop( const toolResults: ToolResultMessage[] = []; if (hasMoreToolCalls) { - const toolExecution = await executeToolCalls( + const executionResult = await executeToolCalls( currentContext.tools, message, signal, stream, config.getSteeringMessages, - config.getToolContext, config.interruptMode, + config.getToolContext, config.transformToolCallArguments, config.intentTracing, ); - toolResults.push(...toolExecution.toolResults); - steeringAfterTools = toolExecution.steeringMessages ?? null; + + toolResults.push(...executionResult.toolResults); + steeringMessagesFromExecution = executionResult.steeringMessages; for (const result of toolResults) { currentContext.messages.push(result); @@ -272,13 +273,7 @@ async function runLoop( stream.push({ type: "turn_end", message, toolResults }); - // Get steering messages after turn completes - if (steeringAfterTools && steeringAfterTools.length > 0) { - pendingMessages = steeringAfterTools; - steeringAfterTools = null; - } else { - pendingMessages = (await config.getSteeringMessages?.()) || []; - } + pendingMessages = steeringMessagesFromExecution ?? ((await config.getSteeringMessages?.()) || []); } // Agent would stop here. Check for follow-up messages. @@ -433,25 +428,37 @@ async function executeToolCalls( signal: AbortSignal | undefined, stream: EventStream, getSteeringMessages?: AgentLoopConfig["getSteeringMessages"], - getToolContext?: AgentLoopConfig["getToolContext"], interruptMode: AgentLoopConfig["interruptMode"] = "immediate", + getToolContext?: AgentLoopConfig["getToolContext"], transformToolCallArguments?: AgentLoopConfig["transformToolCallArguments"], intentTracing?: AgentLoopConfig["intentTracing"], ): Promise<{ toolResults: ToolResultMessage[]; steeringMessages?: AgentMessage[] }> { type ToolCallContent = Extract; const toolCalls = assistantMessage.content.filter((c): c is ToolCallContent => c.type === "toolCall"); const emittedToolResults: ToolResultMessage[] = []; - let steeringMessages: AgentMessage[] | undefined; - const shouldInterruptImmediately = interruptMode !== "wait"; const toolCallInfos = toolCalls.map(call => ({ id: call.id, name: call.name })); const batchId = `${assistantMessage.timestamp ?? Date.now()}_${toolCalls[0]?.id ?? "batch"}`; + const shouldInterruptImmediately = interruptMode !== "wait"; const steeringAbortController = new AbortController(); const toolSignal = signal ? AbortSignal.any([signal, steeringAbortController.signal]) : steeringAbortController.signal; const interruptState = { triggered: false }; + let steeringMessages: AgentMessage[] | undefined; let steeringCheck: Promise | null = null; + const records = toolCalls.map(toolCall => ({ + toolCall, + tool: tools?.find(t => t.name === toolCall.name), + args: toolCall.arguments as Record, + started: false, + result: undefined as AgentToolResult | undefined, + isError: false, + skipped: false, + toolResultMessage: undefined as ToolResultMessage | undefined, + resultEmitted: false, + })); + const checkSteering = async (): Promise => { if (!shouldInterruptImmediately || !getSteeringMessages || interruptState.triggered) { return; @@ -473,18 +480,6 @@ async function executeToolCalls( await steeringCheck; }; - const records = toolCalls.map(toolCall => ({ - toolCall, - tool: tools?.find(t => t.name === toolCall.name), - args: toolCall.arguments as Record, - started: false, - result: undefined as AgentToolResult | undefined, - isError: false, - skipped: false, - toolResultMessage: undefined as ToolResultMessage | undefined, - resultEmitted: false, - })); - const emitToolResult = (record: (typeof records)[number], result: AgentToolResult, isError: boolean): void => { if (record.resultEmitted) return; const { toolCall } = record; @@ -578,7 +573,6 @@ async function executeToolCalls( transformToolCallArguments ? transformToolCallArguments(effectiveArgs, toolCall.name) : effectiveArgs, tool.nonAbortable ? undefined : toolSignal, partialResult => { - if (interruptState.triggered) return; stream.push({ type: "tool_execution_update", toolCallId: toolCall.id, @@ -637,13 +631,6 @@ async function executeToolCalls( return { toolResults: emittedToolResults, steeringMessages }; } -function createSkippedToolResult(): AgentToolResult { - return { - content: [{ type: "text", text: "Skipped due to queued user message." }], - details: {}, - }; -} - /** * Create a tool result for a tool call that was aborted or errored before execution. * Maintains the tool_use/tool_result pairing required by the API. @@ -690,3 +677,10 @@ function createAbortedToolResult( return toolResultMessage; } + +function createSkippedToolResult(): AgentToolResult { + return { + content: [{ type: "text", text: "Skipped due to queued user message." }], + details: {}, + }; +} diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index ba73a7f81..da002c77b 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -603,33 +603,16 @@ describe("agentLoop with AgentMessage", () => { expect(text).not.toContain("Tool execution was aborted.:"); } }); - it("should inject queued messages and skip remaining tool calls", async () => { + it("should skip remaining tool calls when steering is queued", async () => { const toolSchema = Type.Object({ value: Type.String() }); const executed: string[] = []; - const { promise: allowSecond, resolve: allowSecondResolve } = Promise.withResolvers(); const tool: AgentTool = { name: "echo", label: "Echo", description: "Echo tool", parameters: toolSchema, - async execute(_toolCallId, params, signal) { - if (params.value === "second") { - await new Promise((resolve, reject) => { - if (signal?.aborted) { - reject(new Error("Tool aborted")); - return; - } - const onAbort = () => reject(new Error("Tool aborted")); - signal?.addEventListener("abort", onAbort, { once: true }); - allowSecond.then(() => { - signal?.removeEventListener("abort", onAbort); - resolve(); - }); - }); - if (signal?.aborted) { - throw new Error("Tool aborted"); - } - } + concurrency: "exclusive", + async execute(_toolCallId, params) { executed.push(params.value); return { content: [{ type: "text", text: `ok:${params.value}` }], @@ -654,11 +637,11 @@ describe("agentLoop with AgentMessage", () => { const config: AgentLoopConfig = { model: createModel(), convertToLlm: identityConverter, + interruptMode: "immediate", getSteeringMessages: async () => { - // Return queued message after first tool executes - if (executed.length === 1 && !queuedDelivered) { + // Return steering message after tool execution has started + if (executed.length >= 1 && !queuedDelivered) { queuedDelivered = true; - allowSecondResolve(); return [queuedUserMessage]; } return []; @@ -700,29 +683,31 @@ describe("agentLoop with AgentMessage", () => { events.push(event); } - // Only first tool should have executed + // Only the first tool should execute; the second is skipped after steering is queued. expect(executed).toEqual(["first"]); - // Second tool should be skipped const toolEnds = events.filter( (e): e is Extract => e.type === "tool_execution_end", ); expect(toolEnds.length).toBe(2); - expect(toolEnds[0].isError).toBeFalsy(); + expect(toolEnds[0].isError).toBe(false); expect(toolEnds[1].isError).toBe(true); if (toolEnds[1].result.content[0]?.type === "text") { expect(toolEnds[1].result.content[0].text).toContain("Skipped due to queued user message"); } - // Queued message should appear in events - const queuedMessageEvent = events.find( - e => - e.type === "message_start" && - e.message.role === "user" && - typeof e.message.content === "string" && - e.message.content === "interrupt", - ); - expect(queuedMessageEvent).toBeDefined(); + // Queued message should appear in events after the tool results and before the next model call. + const eventSequence = events.flatMap(event => { + if (event.type !== "message_start") return []; + if (event.message.role === "toolResult") return [`tool:${event.message.toolCallId}`]; + if (event.message.role === "user" && typeof event.message.content === "string") { + return [event.message.content]; + } + return []; + }); + expect(eventSequence).toContain("interrupt"); + expect(eventSequence.indexOf("tool:tool-1")).toBeLessThan(eventSequence.indexOf("interrupt")); + expect(eventSequence.indexOf("tool:tool-2")).toBeLessThan(eventSequence.indexOf("interrupt")); // Interrupt message should be in context when second LLM call is made expect(sawInterruptInContext).toBe(true); diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index df719797a..7acc629dd 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -16,6 +16,7 @@ export * from "./providers/google"; export * from "./providers/google-gemini-cli"; export * from "./providers/google-vertex"; export * from "./providers/kimi"; +export type { OpenAICodexResponsesOptions } from "./providers/openai-codex-responses"; export * from "./providers/openai-completions"; export * from "./providers/openai-responses"; export * from "./providers/synthetic"; diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index a036c5e79..d33a3fb99 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -360,6 +360,13 @@ function handleContentBlockStop( /** * Check if the model supports prompt caching. * Supported: Claude 3.5 Haiku, Claude 3.7 Sonnet, Claude 4.x+ models, Haiku 4.5+ + * + * For base models and system-defined inference profiles the model ID / ARN + * contains the model name, so we can decide locally. + * + * For application inference profiles (whose ARNs don't contain the model name), + * set AWS_BEDROCK_FORCE_CACHE=1 to enable cache points. Amazon Nova models + * have automatic caching and don't need explicit cache points. */ function supportsPromptCaching(model: Model<"bedrock-converse-stream">): boolean { if (model.cost.cacheRead || model.cost.cacheWrite) return true; @@ -370,6 +377,9 @@ function supportsPromptCaching(model: Model<"bedrock-converse-stream">): boolean if (id.includes("claude-3-7-sonnet") || id.includes("claude-3-5-haiku")) return true; // Claude Haiku 4.5+ (new naming) if (id.includes("claude-haiku")) return true; + // Application inference profiles don't contain the model name in the ARN. + // Allow users to force cache points via environment variable. + if (typeof process !== "undefined" && process.env.AWS_BEDROCK_FORCE_CACHE === "1") return true; return false; } diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 9d409a5c4..3d42663f7 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -376,6 +376,12 @@ export interface AnthropicOptions extends StreamOptions { betas?: string[] | string; /** Force OAuth bearer auth mode for proxy tokens that don't match Anthropic token prefixes. */ isOAuth?: boolean; + /** + * Pre-built Anthropic client instance. When provided, skips internal client + * construction entirely. Use this to inject alternative SDK clients such as + * `AnthropicVertex` that shares the same messaging API. + */ + client?: Anthropic; } export type AnthropicClientOptionsArgs = { @@ -611,19 +617,31 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( let rawRequestDump: RawHttpRequestDump | undefined; try { - const apiKey = options?.apiKey ?? getEnvApiKey(model.provider) ?? ""; - const baseUrl = resolveAnthropicBaseUrl(model, apiKey) ?? "https://api.anthropic.com"; + let client: Anthropic; + let isOAuthToken: boolean; - const { client, isOAuthToken } = createClient(model, { - model, - apiKey, - extraBetas: normalizeExtraBetas(options?.betas), - stream: true, - interleavedThinking: options?.interleavedThinking ?? true, - headers: options?.headers, - dynamicHeaders: copilotDynamicHeaders?.headers, - isOAuth: options?.isOAuth, - }); + if (options?.client) { + client = options.client; + isOAuthToken = false; + } else { + const apiKey = options?.apiKey ?? getEnvApiKey(model.provider) ?? ""; + + const created = createClient(model, { + model, + apiKey, + extraBetas: normalizeExtraBetas(options?.betas), + stream: true, + interleavedThinking: options?.interleavedThinking ?? true, + headers: options?.headers, + dynamicHeaders: copilotDynamicHeaders?.headers, + isOAuth: options?.isOAuth, + }); + client = created.client; + isOAuthToken = created.isOAuthToken; + } + const baseUrl = + resolveAnthropicBaseUrl(model, options?.apiKey ?? getEnvApiKey(model.provider) ?? "") ?? + "https://api.anthropic.com"; let params = buildParams(model, baseUrl, context, isOAuthToken, options); const replacementPayload = await options?.onPayload?.(params, model); if (replacementPayload !== undefined) { @@ -661,6 +679,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( for await (const event of anthropicStream) { started = true; if (event.type === "message_start") { + output.responseId = event.message.id; // Capture initial token usage from message_start event // This ensures we have input token counts even if the stream is aborted early output.usage.input = event.message.usage.input_tokens || 0; diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index 5a524486a..595aa29f0 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -68,6 +68,20 @@ export function requiresToolCallId(modelId: string): boolean { return modelId.startsWith("claude-"); } +function getGeminiMajorVersion(modelId: string): number | undefined { + const match = modelId.toLowerCase().match(/^gemini(?:-live)?-(\d+)/); + if (!match) return undefined; + return Number.parseInt(match[1], 10); +} + +function supportsMultimodalFunctionResponse(modelId: string): boolean { + const geminiMajorVersion = getGeminiMajorVersion(modelId); + if (geminiMajorVersion !== undefined) { + return geminiMajorVersion >= 3; + } + return true; +} + function isGemini3Model(modelId: string): boolean { return modelId.includes("gemini-3"); } @@ -189,10 +203,10 @@ export function convertMessages(model: Model, contex const hasText = textResult.length > 0; const hasImages = imageContent.length > 0; - // Gemini 3 supports multimodal function responses with images nested inside functionResponse.parts - // See: https://ai.google.dev/gemini-api/docs/function-calling#multimodal - // Older models don't support this, so we put images in a separate user message. - const supportsMultimodalFunctionResponse = model.id.includes("gemini-3"); + // Gemini 3+ models support multimodal function responses with images nested inside + // functionResponse.parts. Claude and other non-Gemini models behind Cloud Code Assist / + // Antigravity also accept this shape. Gemini < 3 still needs a separate user image turn. + const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id); // Use "output" key for success, "error" key for errors as per SDK documentation const responseValue = hasText ? textResult.toWellFormed() : hasImages ? "(see attached image)" : ""; @@ -209,8 +223,7 @@ export function convertMessages(model: Model, contex functionResponse: { name: msg.toolName, response: msg.isError ? { error: responseValue } : { output: responseValue }, - // Nest images inside functionResponse.parts for Gemini 3 - ...(hasImages && supportsMultimodalFunctionResponse && { parts: imageParts }), + ...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }), ...(includeId ? { id: msg.toolCallId } : {}), }, }; @@ -231,8 +244,8 @@ export function convertMessages(model: Model, contex }); } - // For older models, add images in a separate user message - if (hasImages && !supportsMultimodalFunctionResponse) { + // For Gemini < 3, add images in a separate user message + if (hasImages && !modelSupportsMultimodalFunctionResponse) { contents.push({ role: "user", parts: [{ text: "Tool result image:" }, ...imageParts], diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index f3f44b50b..fc5ace911 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -811,7 +811,7 @@ function handleCodexStreamEvent(args: { return handleResponseCreated(runtime, rawEvent); } - if (eventType === "response.completed" || eventType === "response.done") { + if (eventType === "response.completed" || eventType === "response.done" || eventType === "response.incomplete") { handleResponseCompleted(model, output, runtime, rawEvent); return firstTokenTime; } @@ -1046,6 +1046,9 @@ function handleResponseCompleted( cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, }; } + if (typeof response?.id === "string" && response.id.length > 0) { + output.responseId = response.id; + } const state = runtime.websocketState; if (runtime.transport === "websocket" && state) { @@ -1764,6 +1767,7 @@ class CodexWebSocketConnection { if ( eventType === "response.completed" || eventType === "response.done" || + eventType === "response.incomplete" || eventType === "response.failed" || eventType === "error" ) { diff --git a/packages/ai/src/providers/openai-completions-compat.ts b/packages/ai/src/providers/openai-completions-compat.ts index f75ce4b47..d6ff51bcb 100644 --- a/packages/ai/src/providers/openai-completions-compat.ts +++ b/packages/ai/src/providers/openai-completions-compat.ts @@ -89,7 +89,13 @@ export function detectOpenAICompat(model: Model<"openai-completions">): Resolved requiresAssistantAfterToolResult: false, requiresThinkingAsText: isMistral, requiresMistralToolIds: isMistral, - thinkingFormat: isZai ? "zai" : isAlibaba || isQwen ? "qwen" : "openai", + thinkingFormat: isZai + ? "zai" + : provider === "openrouter" || baseUrl.includes("openrouter.ai") + ? "openrouter" + : isAlibaba || isQwen + ? "qwen" + : "openai", reasoningContentField: "reasoning_content", requiresReasoningContentForToolCalls: isKimiModel, requiresAssistantContentForToolCalls: isKimiModel, diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index cd8e12bc8..f26545224 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -337,11 +337,17 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( errorMessage: "OpenAI completions stream stalled while waiting for the next event", onIdle: () => requestAbortController.abort(), })) { + if (!chunk || typeof chunk !== "object") continue; + + // OpenAI documents ChatCompletionChunk.id as the unique chat completion identifier, + // and each chunk in a streamed completion carries the same id. + output.responseId ||= chunk.id; + if (chunk.usage) { output.usage = parseChunkUsage(chunk.usage, model, copilotPremiumRequests); } - const choice = chunk.choices[0]; + const choice = Array.isArray(chunk.choices) ? chunk.choices[0] : undefined; if (!choice) continue; if (!chunk.usage) { @@ -352,7 +358,11 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( } if (choice.finish_reason) { - output.stopReason = mapStopReason(choice.finish_reason); + const finishReasonResult = mapStopReason(choice.finish_reason); + output.stopReason = finishReasonResult.stopReason; + if (finishReasonResult.errorMessage) { + output.errorMessage = finishReasonResult.errorMessage; + } } if (choice.delta) { @@ -463,8 +473,11 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( throw new Error("Request was aborted"); } - if (output.stopReason === "aborted" || output.stopReason === "error") { - throw new Error("An unknown error occurred"); + if (output.stopReason === "aborted") { + throw new Error("Request was aborted"); + } + if (output.stopReason === "error") { + throw new Error(output.errorMessage || "Provider returned an error stop reason"); } output.duration = Date.now() - startTime; @@ -616,6 +629,12 @@ function buildParams(model: Model<"openai-completions">, context: Context, optio Reflect.set(params, "enable_thinking", !!options?.reasoning); } else if (compat.thinkingFormat === "qwen-chat-template" && model.reasoning) { Reflect.set(params, "chat_template_kwargs", { enable_thinking: !!options?.reasoning }); + } else if (compat.thinkingFormat === "openrouter" && options?.reasoning && model.reasoning) { + // OpenRouter normalizes reasoning across providers via a nested reasoning object. + const openRouterParams = params as typeof params & { reasoning?: { effort?: string } }; + openRouterParams.reasoning = { + effort: mapReasoningEffort(options.reasoning, compat.reasoningEffortMap), + }; } else if (options?.reasoning && model.reasoning && compat.supportsReasoningEffort) { // OpenAI-style reasoning_effort Reflect.set(params, "reasoning_effort", mapReasoningEffort(options.reasoning, compat.reasoningEffortMap)); @@ -1061,21 +1080,29 @@ function convertTools(tools: Tool[], compat: ResolvedOpenAICompat): OpenAI.Chat. }); } -function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"] | string): StopReason { - if (reason === null) return "stop"; +function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"] | string): { + stopReason: StopReason; + errorMessage?: string; +} { + if (reason === null) return { stopReason: "stop" }; switch (reason) { case "stop": case "end": - return "stop"; + return { stopReason: "stop" }; case "length": - return "length"; + return { stopReason: "length" }; case "function_call": case "tool_calls": - return "toolUse"; + return { stopReason: "toolUse" }; case "content_filter": - return "error"; + return { stopReason: "error", errorMessage: "Provider finish_reason: content_filter" }; + case "network_error": + return { stopReason: "error", errorMessage: "Provider finish_reason: network_error" }; default: - throw new Error(`Unhandled stop reason: ${reason}`); + return { + stopReason: "error", + errorMessage: `Provider finish_reason: ${reason}`, + }; } } diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 3ef822262..74341fcc7 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -52,8 +52,26 @@ export function parseTextSignature( return { id: signature }; } -export function normalizeResponsesToolCallIdForTransform(id: string): string { +export function normalizeResponsesToolCallIdForTransform( + id: string, + model?: Model, + source?: AssistantMessage, +): string { if (!id.includes("|")) return id; + const isForeignToolCall = + source != null && model != null && (source.provider !== model.provider || source.api !== model.api); + if (isForeignToolCall) { + const [callId, itemId] = id.split("|"); + const normalizeIdPart = (part: string): string => { + const sanitized = part.replace(/[^a-zA-Z0-9_-]/g, "_"); + const truncated = sanitized.length > 64 ? sanitized.slice(0, 64) : sanitized; + return truncated.replace(/_+$/, ""); + }; + const normalizedCallId = normalizeIdPart(callId); + let normalizedItemId = `fc_${Bun.hash(itemId).toString(36)}`; + if (normalizedItemId.length > 64) normalizedItemId = normalizedItemId.slice(0, 64); + return `${normalizedCallId}|${normalizedItemId}`; + } const normalized = normalizeResponsesToolCallId(id); return `${normalized.callId}|${normalized.itemId}`; } @@ -221,7 +239,9 @@ export async function processResponsesStream( let sawFirstToken = false; for await (const event of openaiStream) { - if (event.type === "response.output_item.added") { + if (event.type === "response.created") { + output.responseId = event.response.id; + } else if (event.type === "response.output_item.added") { if (!sawFirstToken) { sawFirstToken = true; options?.onFirstToken?.(); @@ -376,6 +396,9 @@ export async function processResponsesStream( } } else if (event.type === "response.completed") { const response = event.response; + if (response?.id) { + output.responseId = response.id; + } if (response?.usage) { const cachedTokens = response.usage.input_tokens_details?.cached_tokens || 0; output.usage = { diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts new file mode 100644 index 000000000..ab7ba9eef --- /dev/null +++ b/packages/ai/src/providers/register-builtins.ts @@ -0,0 +1,309 @@ +/** + * Lazy provider module loading. + * + * Each provider module is loaded only when its stream function is first called. + * This avoids eagerly importing heavy SDK dependencies (e.g., @anthropic-ai/sdk, + * openai) at startup. The loaded module promise is cached so subsequent calls + * reuse the same import. + * + * NOTE: stream.ts currently imports providers directly, so this file is not yet + * wired into the main streaming path. It provides the infrastructure for lazy + * loading that can be integrated when stream.ts is refactored. + */ +import type { + Api, + AssistantMessage, + AssistantMessageEvent, + AssistantMessageEventStream, + Context, + Model, + OptionsForApi, +} from "../types"; +import { AssistantMessageEventStream as EventStreamImpl } from "../utils/event-stream"; +import type { BedrockOptions } from "./amazon-bedrock"; +import type { AnthropicOptions } from "./anthropic"; +import type { AzureOpenAIResponsesOptions } from "./azure-openai-responses"; +import type { CursorOptions } from "./cursor"; +import type { GoogleOptions } from "./google"; +import type { GoogleGeminiCliOptions } from "./google-gemini-cli"; +import type { GoogleVertexOptions } from "./google-vertex"; +import type { OpenAICodexResponsesOptions } from "./openai-codex-responses"; +import type { OpenAICompletionsOptions } from "./openai-completions"; +import type { OpenAIResponsesOptions } from "./openai-responses"; + +// --------------------------------------------------------------------------- +// Lazy provider module shape +// --------------------------------------------------------------------------- + +interface LazyProviderModule { + stream: (model: Model, context: Context, options: OptionsForApi) => AsyncIterable; +} + +interface AnthropicProviderModule { + streamAnthropic: ( + model: Model<"anthropic-messages">, + context: Context, + options: AnthropicOptions, + ) => AssistantMessageEventStream; +} + +interface AzureOpenAIResponsesProviderModule { + streamAzureOpenAIResponses: ( + model: Model<"azure-openai-responses">, + context: Context, + options: AzureOpenAIResponsesOptions, + ) => AssistantMessageEventStream; +} + +interface GoogleProviderModule { + streamGoogle: ( + model: Model<"google-generative-ai">, + context: Context, + options: GoogleOptions, + ) => AssistantMessageEventStream; +} + +interface GoogleGeminiCliProviderModule { + streamGoogleGeminiCli: ( + model: Model<"google-gemini-cli">, + context: Context, + options: GoogleGeminiCliOptions, + ) => AssistantMessageEventStream; +} + +interface GoogleVertexProviderModule { + streamGoogleVertex: ( + model: Model<"google-vertex">, + context: Context, + options: GoogleVertexOptions, + ) => AssistantMessageEventStream; +} + +interface OpenAICodexResponsesProviderModule { + streamOpenAICodexResponses: ( + model: Model<"openai-codex-responses">, + context: Context, + options: OpenAICodexResponsesOptions, + ) => AssistantMessageEventStream; +} + +interface OpenAICompletionsProviderModule { + streamOpenAICompletions: ( + model: Model<"openai-completions">, + context: Context, + options: OpenAICompletionsOptions, + ) => AssistantMessageEventStream; +} + +interface OpenAIResponsesProviderModule { + streamOpenAIResponses: ( + model: Model<"openai-responses">, + context: Context, + options: OpenAIResponsesOptions, + ) => AssistantMessageEventStream; +} + +interface CursorProviderModule { + streamCursor: ( + model: Model<"cursor-agent">, + context: Context, + options: CursorOptions, + ) => AssistantMessageEventStream; +} + +interface BedrockProviderModule { + streamBedrock: ( + model: Model<"bedrock-converse-stream">, + context: Context, + options: BedrockOptions, + ) => AssistantMessageEventStream; +} + +// --------------------------------------------------------------------------- +// Module-level lazy promise caches +// --------------------------------------------------------------------------- + +const importNodeOnlyProvider = (specifier: string): Promise => import(specifier); + +let anthropicProviderModulePromise: Promise> | undefined; +let azureOpenAIResponsesProviderModulePromise: Promise> | undefined; +let googleProviderModulePromise: Promise> | undefined; +let googleGeminiCliProviderModulePromise: Promise> | undefined; +let googleVertexProviderModulePromise: Promise> | undefined; +let openAICodexResponsesProviderModulePromise: Promise> | undefined; +let openAICompletionsProviderModulePromise: Promise> | undefined; +let openAIResponsesProviderModulePromise: Promise> | undefined; +let cursorProviderModulePromise: Promise> | undefined; +let bedrockProviderModuleOverride: LazyProviderModule<"bedrock-converse-stream"> | undefined; +let bedrockProviderModulePromise: Promise> | undefined; + +export function setBedrockProviderModule(module: BedrockProviderModule): void { + bedrockProviderModuleOverride = { + stream: module.streamBedrock, + }; +} + +// --------------------------------------------------------------------------- +// Stream forwarding / error helpers +// --------------------------------------------------------------------------- + +function forwardStream(target: EventStreamImpl, source: AsyncIterable): void { + (async () => { + for await (const event of source) { + target.push(event); + } + target.end(); + })(); +} + +function createLazyLoadErrorMessage(model: Model, error: unknown): AssistantMessage { + return { + role: "assistant", + content: [], + api: model.api, + provider: model.provider, + model: model.id, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "error", + errorMessage: error instanceof Error ? error.message : String(error), + timestamp: Date.now(), + }; +} + +// --------------------------------------------------------------------------- +// Generic lazy stream factory +// --------------------------------------------------------------------------- + +function createLazyStream( + loadModule: () => Promise>, +): (model: Model, context: Context, options: OptionsForApi) => EventStreamImpl { + return (model, context, options) => { + const outer = new EventStreamImpl(); + + loadModule() + .then(module => { + const inner = module.stream(model, context, options); + forwardStream(outer, inner); + }) + .catch(error => { + const message = createLazyLoadErrorMessage(model, error); + outer.push({ type: "error", reason: "error", error: message }); + outer.end(message); + }); + + return outer; + }; +} + +// --------------------------------------------------------------------------- +// Module loaders (one per provider, cached via ||=) +// --------------------------------------------------------------------------- + +function loadAnthropicProviderModule(): Promise> { + anthropicProviderModulePromise ||= import("./anthropic").then(module => { + const provider = module as AnthropicProviderModule; + return { stream: provider.streamAnthropic }; + }); + return anthropicProviderModulePromise; +} + +function loadAzureOpenAIResponsesProviderModule(): Promise> { + azureOpenAIResponsesProviderModulePromise ||= import("./azure-openai-responses").then(module => { + const provider = module as AzureOpenAIResponsesProviderModule; + return { stream: provider.streamAzureOpenAIResponses }; + }); + return azureOpenAIResponsesProviderModulePromise; +} + +function loadGoogleProviderModule(): Promise> { + googleProviderModulePromise ||= import("./google").then(module => { + const provider = module as GoogleProviderModule; + return { stream: provider.streamGoogle }; + }); + return googleProviderModulePromise; +} + +function loadGoogleGeminiCliProviderModule(): Promise> { + googleGeminiCliProviderModulePromise ||= import("./google-gemini-cli").then(module => { + const provider = module as GoogleGeminiCliProviderModule; + return { stream: provider.streamGoogleGeminiCli }; + }); + return googleGeminiCliProviderModulePromise; +} + +function loadGoogleVertexProviderModule(): Promise> { + googleVertexProviderModulePromise ||= import("./google-vertex").then(module => { + const provider = module as GoogleVertexProviderModule; + return { stream: provider.streamGoogleVertex }; + }); + return googleVertexProviderModulePromise; +} + +function loadOpenAICodexResponsesProviderModule(): Promise> { + openAICodexResponsesProviderModulePromise ||= import("./openai-codex-responses").then(module => { + const provider = module as OpenAICodexResponsesProviderModule; + return { stream: provider.streamOpenAICodexResponses }; + }); + return openAICodexResponsesProviderModulePromise; +} + +function loadOpenAICompletionsProviderModule(): Promise> { + openAICompletionsProviderModulePromise ||= import("./openai-completions").then(module => { + const provider = module as OpenAICompletionsProviderModule; + return { stream: provider.streamOpenAICompletions }; + }); + return openAICompletionsProviderModulePromise; +} + +function loadOpenAIResponsesProviderModule(): Promise> { + openAIResponsesProviderModulePromise ||= import("./openai-responses").then(module => { + const provider = module as OpenAIResponsesProviderModule; + return { stream: provider.streamOpenAIResponses }; + }); + return openAIResponsesProviderModulePromise; +} + +function loadCursorProviderModule(): Promise> { + cursorProviderModulePromise ||= import("./cursor").then(module => { + const provider = module as CursorProviderModule; + return { stream: provider.streamCursor }; + }); + return cursorProviderModulePromise; +} + +function loadBedrockProviderModule(): Promise> { + if (bedrockProviderModuleOverride) { + return Promise.resolve(bedrockProviderModuleOverride); + } + bedrockProviderModulePromise ||= importNodeOnlyProvider("./amazon-bedrock").then(module => { + const provider = module as BedrockProviderModule; + return { stream: provider.streamBedrock }; + }); + return bedrockProviderModulePromise; +} + +// --------------------------------------------------------------------------- +// Lazy stream function exports +// +// These use the same names as the direct provider stream functions. When +// stream.ts is updated to import from this module instead of individual +// providers, the lazy loading will take effect on the main code path. +// --------------------------------------------------------------------------- + +export const streamAnthropic = createLazyStream(loadAnthropicProviderModule); +export const streamAzureOpenAIResponses = createLazyStream(loadAzureOpenAIResponsesProviderModule); +export const streamGoogle = createLazyStream(loadGoogleProviderModule); +export const streamGoogleGeminiCli = createLazyStream(loadGoogleGeminiCliProviderModule); +export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule); +export const streamOpenAICodexResponses = createLazyStream(loadOpenAICodexResponsesProviderModule); +export const streamOpenAICompletions = createLazyStream(loadOpenAICompletionsProviderModule); +export const streamOpenAIResponses = createLazyStream(loadOpenAIResponsesProviderModule); +export const streamCursor = createLazyStream(loadCursorProviderModule); +export const streamBedrock = createLazyStream(loadBedrockProviderModule); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index de610c366..e38224e60 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -338,6 +338,7 @@ export interface AssistantMessage { api: Api; provider: Provider; model: string; + responseId?: string; // Provider-specific response/message identifier when the upstream API exposes one usage: Usage; stopReason: StopReason; errorMessage?: string; @@ -444,8 +445,8 @@ export interface OpenAICompat { requiresThinkingAsText?: boolean; /** Whether tool call IDs must be normalized to Mistral format (exactly 9 alphanumeric chars). Default: auto-detected from URL. */ requiresMistralToolIds?: boolean; - /** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "zai" uses thinking: { type: "enabled" }, "qwen" uses top-level enable_thinking, and "qwen-chat-template" uses chat_template_kwargs.enable_thinking. Default: "openai". */ - thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template"; + /** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "zai" uses thinking: { type: "enabled" }, "qwen" uses top-level enable_thinking, and "qwen-chat-template" uses chat_template_kwargs.enable_thinking. Default: "openai". */ + thinkingFormat?: "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template"; /** Which reasoning content field to emit on assistant messages. Default: auto-detected. */ reasoningContentField?: "reasoning_content" | "reasoning" | "reasoning_text"; /** Whether assistant tool-call messages must include reasoning content. Default: false. */ diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 3b5b1392e..751dd271b 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -28,6 +28,7 @@ export interface Args { mode?: Mode; noSession?: boolean; sessionDir?: string; + fork?: string; models?: string[]; tools?: string[]; noTools?: boolean; @@ -79,6 +80,8 @@ export function parseArgs(args: string[], extensionFlags?: Map 0; + if (!hasInitialContext) { + return { + initialImages: undefined, + }; + } + + let body = ""; + if (fileText !== undefined) { + body += fileText; + } + + if (parsed.messages.length > 0) { + body += parsed.messages[0]; + parsed.messages.shift(); + } + + const initialMessage = + stdinContent !== undefined + ? body.length > 0 + ? `${stdinContent}\n${body}` + : stdinContent + : body.length > 0 + ? body + : fileImages && fileImages.length > 0 + ? "" + : undefined; + + return { + initialMessage, + initialImages: fileImages && fileImages.length > 0 ? fileImages : undefined, + }; +} diff --git a/packages/coding-agent/src/config/keybindings.ts b/packages/coding-agent/src/config/keybindings.ts index 6fe4ee578..44d954e4a 100644 --- a/packages/coding-agent/src/config/keybindings.ts +++ b/packages/coding-agent/src/config/keybindings.ts @@ -1,120 +1,438 @@ +import { existsSync, readFileSync, writeFileSync } from "node:fs"; import * as path from "node:path"; import { - DEFAULT_EDITOR_KEYBINDINGS, - type EditorAction, - type EditorKeybindingsConfig, - EditorKeybindingsManager, + type Keybinding, + type KeybindingDefinitions, + type KeybindingsConfig, type KeyId, - matchesKey, - setEditorKeybindings, + setKeybindings, + TUI_KEYBINDINGS, + KeybindingsManager as TuiKeybindingsManager, } from "@oh-my-pi/pi-tui"; import { getAgentDir, isEnoent, logger } from "@oh-my-pi/pi-utils"; /** - * Application-level actions (coding agent specific). + * Application-level keybindings (coding agent specific). + * Values are always `true` — used for declaration merging. */ -export type AppAction = - | "interrupt" - | "clear" - | "exit" - | "suspend" - | "cycleThinkingLevel" - | "cycleModelForward" - | "cycleModelBackward" - | "selectModel" - | "togglePlanMode" - | "expandTools" - | "toggleThinking" - | "externalEditor" - | "historySearch" - | "followUp" - | "dequeue" - | "pasteImage" - | "copyLine" - | "copyPrompt" - | "newSession" - | "tree" - | "fork" - | "resume" - | "toggleSTT"; +interface AppKeybindings { + "app.interrupt": true; + "app.clear": true; + "app.exit": true; + "app.suspend": true; + "app.thinking.cycle": true; + "app.thinking.toggle": true; + "app.model.cycleForward": true; + "app.model.cycleBackward": true; + "app.model.select": true; + "app.tools.expand": true; + "app.editor.external": true; + "app.message.followUp": true; + "app.message.dequeue": true; + "app.clipboard.pasteImage": true; + "app.clipboard.copyLine": true; + "app.clipboard.copyPrompt": true; + "app.session.new": true; + "app.session.tree": true; + "app.session.fork": true; + "app.session.resume": true; + "app.session.togglePath": true; + "app.session.toggleSort": true; + "app.session.rename": true; + "app.session.delete": true; + "app.session.deleteNoninvasive": true; + "app.tree.foldOrUp": true; + "app.tree.unfoldOrDown": true; + "app.plan.toggle": true; + "app.history.search": true; + "app.stt.toggle": true; +} + +export type AppKeybinding = keyof AppKeybindings; + +declare module "@oh-my-pi/pi-tui" { + interface Keybindings extends AppKeybindings {} +} /** - * All configurable actions. + * All keybindings definitions: TUI + app-specific. */ -export type KeyAction = AppAction | EditorAction; +export const KEYBINDINGS = { + ...TUI_KEYBINDINGS, + "app.interrupt": { + defaultKeys: "escape", + description: "Interrupt current operation", + }, + "app.clear": { + defaultKeys: "ctrl+c", + description: "Clear screen or cancel", + }, + "app.exit": { + defaultKeys: "ctrl+d", + description: "Exit application", + }, + "app.suspend": { + defaultKeys: "ctrl+z", + description: "Suspend application", + }, + "app.thinking.cycle": { + defaultKeys: "shift+tab", + description: "Cycle thinking level", + }, + "app.thinking.toggle": { + defaultKeys: "ctrl+t", + description: "Toggle thinking mode", + }, + "app.model.cycleForward": { + defaultKeys: "ctrl+p", + description: "Cycle to next model", + }, + "app.model.cycleBackward": { + defaultKeys: "shift+ctrl+p", + description: "Cycle to previous model", + }, + "app.model.select": { + defaultKeys: "ctrl+l", + description: "Select model", + }, + "app.tools.expand": { + defaultKeys: "ctrl+o", + description: "Expand tools", + }, + "app.editor.external": { + defaultKeys: "ctrl+g", + description: "Open external editor", + }, + "app.message.followUp": { + defaultKeys: "ctrl+enter", + description: "Send follow-up message", + }, + "app.message.dequeue": { + defaultKeys: "alt+up", + description: "Dequeue message", + }, + "app.clipboard.pasteImage": { + defaultKeys: process.platform === "win32" ? "alt+v" : "ctrl+v", + description: "Paste image from clipboard", + }, + "app.clipboard.copyLine": { + defaultKeys: "alt+shift+l", + description: "Copy current line", + }, + "app.clipboard.copyPrompt": { + defaultKeys: "alt+shift+c", + description: "Copy prompt", + }, + "app.session.new": { + defaultKeys: [], + description: "Create new session", + }, + "app.session.tree": { + defaultKeys: [], + description: "Show session tree", + }, + "app.session.fork": { + defaultKeys: [], + description: "Fork session", + }, + "app.session.resume": { + defaultKeys: [], + description: "Resume session", + }, + "app.session.togglePath": { + defaultKeys: "ctrl+p", + description: "Toggle session path display", + }, + "app.session.toggleSort": { + defaultKeys: "ctrl+s", + description: "Toggle session sort order", + }, + "app.session.rename": { + defaultKeys: "ctrl+r", + description: "Rename session", + }, + "app.session.delete": { + defaultKeys: "ctrl+d", + description: "Delete session", + }, + "app.session.deleteNoninvasive": { + defaultKeys: "ctrl+backspace", + description: "Delete session (non-invasive)", + }, + "app.tree.foldOrUp": { + defaultKeys: ["ctrl+left", "alt+left"], + description: "Fold or move up", + }, + "app.tree.unfoldOrDown": { + defaultKeys: ["ctrl+right", "alt+right"], + description: "Unfold or move down", + }, + "app.plan.toggle": { + defaultKeys: "alt+shift+p", + description: "Toggle plan mode", + }, + "app.history.search": { + defaultKeys: "ctrl+r", + description: "Search history", + }, + "app.stt.toggle": { + defaultKeys: "alt+h", + description: "Toggle speech-to-text", + }, +} as const satisfies KeybindingDefinitions; /** - * Full keybindings configuration (app + editor actions). + * Migration map from old keybinding names to new namespaced IDs. */ -export type KeybindingsConfig = { - [K in KeyAction]?: KeyId | KeyId[]; -}; +const KEYBINDING_NAME_MIGRATIONS = { + // App-specific (old names) + interrupt: "app.interrupt", + clear: "app.clear", + exit: "app.exit", + suspend: "app.suspend", + cycleThinkingLevel: "app.thinking.cycle", + cycleModelForward: "app.model.cycleForward", + cycleModelBackward: "app.model.cycleBackward", + selectModel: "app.model.select", + togglePlanMode: "app.plan.toggle", + historySearch: "app.history.search", + expandTools: "app.tools.expand", + toggleThinking: "app.thinking.toggle", + externalEditor: "app.editor.external", + followUp: "app.message.followUp", + dequeue: "app.message.dequeue", + pasteImage: "app.clipboard.pasteImage", + copyLine: "app.clipboard.copyLine", + copyPrompt: "app.clipboard.copyPrompt", + newSession: "app.session.new", + tree: "app.session.tree", + fork: "app.session.fork", + resume: "app.session.resume", + toggleSTT: "app.stt.toggle", + // TUI editor (old names for backward compatibility) + cursorUp: "tui.editor.cursorUp", + cursorDown: "tui.editor.cursorDown", + cursorLeft: "tui.editor.cursorLeft", + cursorRight: "tui.editor.cursorRight", + cursorWordLeft: "tui.editor.cursorWordLeft", + cursorWordRight: "tui.editor.cursorWordRight", + cursorLineStart: "tui.editor.cursorLineStart", + cursorLineEnd: "tui.editor.cursorLineEnd", + jumpForward: "tui.editor.jumpForward", + jumpBackward: "tui.editor.jumpBackward", + pageUp: "tui.editor.pageUp", + pageDown: "tui.editor.pageDown", + deleteCharBackward: "tui.editor.deleteCharBackward", + deleteCharForward: "tui.editor.deleteCharForward", + deleteWordBackward: "tui.editor.deleteWordBackward", + deleteWordForward: "tui.editor.deleteWordForward", + deleteToLineStart: "tui.editor.deleteToLineStart", + deleteToLineEnd: "tui.editor.deleteToLineEnd", + yank: "tui.editor.yank", + yankPop: "tui.editor.yankPop", + undo: "tui.editor.undo", + // TUI input (old names for backward compatibility) + newLine: "tui.input.newLine", + submit: "tui.input.submit", + tab: "tui.input.tab", + copy: "tui.input.copy", + // TUI select (old names for backward compatibility) + selectUp: "tui.select.up", + selectDown: "tui.select.down", + selectPageUp: "tui.select.pageUp", + selectPageDown: "tui.select.pageDown", + selectConfirm: "tui.select.confirm", + selectCancel: "tui.select.cancel", + // Upstream additional migrations + toggleSessionNamedFilter: "app.session.togglePath", +} as const satisfies Record; /** - * Default application keybindings. + * Check if a key is a legacy keybinding name. */ -export const DEFAULT_APP_KEYBINDINGS: Record = { - interrupt: "escape", - clear: "ctrl+c", - exit: "ctrl+d", - suspend: "ctrl+z", - cycleThinkingLevel: "shift+tab", - cycleModelForward: "ctrl+p", - cycleModelBackward: "shift+ctrl+p", - selectModel: "ctrl+l", - togglePlanMode: "alt+shift+p", - historySearch: "ctrl+r", - expandTools: "ctrl+o", - toggleThinking: "ctrl+t", - externalEditor: "ctrl+g", - followUp: "ctrl+enter", - dequeue: "alt+up", - pasteImage: "ctrl+v", - copyLine: "alt+shift+l", - copyPrompt: "alt+shift+c", - newSession: [], - tree: [], - fork: [], - resume: [], - toggleSTT: "alt+h", -}; +function isLegacyKeybindingName(key: string): key is keyof typeof KEYBINDING_NAME_MIGRATIONS { + return key in KEYBINDING_NAME_MIGRATIONS; +} + /** - * All default keybindings (app + editor). + * Normalize input to KeybindingsConfig, validating types. */ -export const DEFAULT_KEYBINDINGS: Required = { - ...DEFAULT_EDITOR_KEYBINDINGS, - ...DEFAULT_APP_KEYBINDINGS, -}; +function toKeybindingsConfig(value: unknown): KeybindingsConfig { + if (typeof value !== "object" || value === null) { + return {}; + } -// App actions list for type checking -const APP_ACTIONS: AppAction[] = [ - "interrupt", - "clear", - "exit", - "suspend", - "cycleThinkingLevel", - "cycleModelForward", - "cycleModelBackward", - "selectModel", - "togglePlanMode", - "historySearch", - "expandTools", - "toggleThinking", - "externalEditor", - "followUp", - "dequeue", - "pasteImage", - "copyLine", - "copyPrompt", - "newSession", - "tree", - "fork", - "resume", - "toggleSTT", -]; + const config: KeybindingsConfig = {}; + for (const [key, val] of Object.entries(value)) { + // Allow undefined, string (KeyId), or array of strings + if (val === undefined) { + config[key] = undefined; + } else if (typeof val === "string") { + config[key] = val as KeyId; + } else if (Array.isArray(val) && val.every(v => typeof v === "string")) { + config[key] = val as string[] as KeyId[]; + } + // Silently skip invalid entries + } + return config; +} -function isAppAction(action: string): action is AppAction { - return APP_ACTIONS.includes(action as AppAction); +/** + * Migrate old keybinding names to new namespaced IDs. + * Returns both the migrated config and a flag indicating if migration occurred. + */ +function migrateKeybindingNames(rawConfig: unknown): { + config: KeybindingsConfig; + migrated: boolean; +} { + const config = toKeybindingsConfig(rawConfig); + const migrated: KeybindingsConfig = {}; + let didMigrate = false; + + for (const [key, value] of Object.entries(config)) { + if (isLegacyKeybindingName(key)) { + const newKey = KEYBINDING_NAME_MIGRATIONS[key]; + migrated[newKey] = value; + didMigrate = true; + } else { + // Already a new-style key + migrated[key] = value; + } + } + + return { config: migrated, migrated: didMigrate }; +} + +/** + * Order keybindings config to match KEYBINDINGS key order. + */ +function orderKeybindingsConfig(config: KeybindingsConfig): KeybindingsConfig { + const ordered: KeybindingsConfig = {}; + for (const key of Object.keys(KEYBINDINGS)) { + const value = config[key]; + if (value !== undefined) { + ordered[key] = value; + } + } + // Add any remaining keys that aren't in KEYBINDINGS + for (const key of Object.keys(config)) { + if (!(key in ordered)) { + ordered[key] = config[key]; + } + } + return ordered; +} + +/** + * Load raw config from a file synchronously. + * Returns parsed JSON or null if file doesn't exist or is invalid. + */ +function loadRawConfig(filePath: string): unknown { + try { + if (!existsSync(filePath)) { + return null; + } + const content = readFileSync(filePath, "utf-8"); + return JSON.parse(content); + } catch (error) { + if (isEnoent(error)) { + return null; + } + logger.warn("Failed to parse keybindings config", { path: filePath, error: String(error) }); + return null; + } +} + +/** + * Migrate keybindings config file from old format to new. + * Reads from agentDir/keybindings.json, migrates old names, and writes back. + */ +function loadKeybindingsConfig(filePath: string, writeBack: boolean): KeybindingsConfig { + const rawConfig = loadRawConfig(filePath); + + if (rawConfig === null) { + return {}; + } + + const { config: migratedConfig, migrated } = migrateKeybindingNames(rawConfig); + if (writeBack && migrated) { + const ordered = orderKeybindingsConfig(migratedConfig); + try { + writeFileSync(filePath, `${JSON.stringify(ordered, null, 2)}\n`, "utf-8"); + logger.debug("Migrated keybindings config", { path: filePath }); + } catch (error) { + logger.warn("Failed to write migrated keybindings config", { path: filePath, error: String(error) }); + } + } + + return migratedConfig; +} + +function migrateKeybindingsConfigFile(agentDir: string): void { + const configPath = path.join(agentDir, "keybindings.json"); + loadKeybindingsConfig(configPath, true); +} + +/** + * Manages all keybindings (app + TUI). + * Extends the TUI KeybindingsManager with app-specific functionality. + */ +export class KeybindingsManager extends TuiKeybindingsManager { + #configPath: string | undefined; + + constructor(userBindings: KeybindingsConfig = {}, configPath?: string) { + super(KEYBINDINGS, userBindings); + this.#configPath = configPath; + } + + /** + * Create from config file at agentDir/keybindings.json. + */ + static create(agentDir: string = getAgentDir()): KeybindingsManager { + const configPath = path.join(agentDir, "keybindings.json"); + const userBindings = KeybindingsManager.#loadFromFile(configPath); + const manager = new KeybindingsManager(userBindings, configPath); + // Set globally so getKeybindings() returns this manager + setKeybindings(manager); + return manager; + } + + /** + * Create an in-memory keybindings manager without file persistence. + */ + static inMemory(userBindings: KeybindingsConfig = {}): KeybindingsManager { + return new KeybindingsManager(userBindings); + } + + /** + * Reload keybindings from the config file. + */ + reload(): void { + if (!this.#configPath) return; + this.setUserBindings(KeybindingsManager.#loadFromFile(this.#configPath)); + } + + /** + * Get the effective resolved bindings (defaults + user overrides). + */ + getEffectiveConfig(): KeybindingsConfig { + return this.getResolvedBindings(); + } + + /** + * Get display string for a keybinding (e.g., "ctrl+c/escape"). + */ + getDisplayString(keybinding: Keybinding): string { + const keys = this.getKeys(keybinding); + return formatKeyHints(keys.length === 0 ? [] : keys); + } + + /** + * Load user bindings from a file, migrating old names if needed. + */ + static #loadFromFile(filePath: string): KeybindingsConfig { + return loadKeybindingsConfig(filePath, true); + } } /** @@ -145,8 +463,6 @@ const KEY_LABELS: Record = { right: "Right", }; -const normalizeKeyId = (key: KeyId): KeyId => key.toLowerCase() as KeyId; - function formatKeyPart(part: string): string { const lower = part.toLowerCase(); const modifier = MODIFIER_LABELS[lower]; @@ -166,116 +482,5 @@ export function formatKeyHints(keys: KeyId | KeyId[]): string { return list.map(formatKeyHint).join("/"); } -/** - * Manages all keybindings (app + editor). - */ -export class KeybindingsManager { - #appActionToKeys: Map; - - private constructor(private readonly config: KeybindingsConfig) { - this.#appActionToKeys = new Map(); - this.#buildMaps(); - } - - /** - * Create from config file and set up editor keybindings. - */ - static async create(agentDir: string = getAgentDir()): Promise { - const configPath = path.join(agentDir, "keybindings.json"); - const config = await KeybindingsManager.#loadFromFile(configPath); - const manager = new KeybindingsManager(config); - - // Set up editor keybindings globally - const editorConfig: EditorKeybindingsConfig = {}; - for (const [action, keys] of Object.entries(config)) { - if (!isAppAction(action)) { - editorConfig[action as EditorAction] = keys; - } - } - setEditorKeybindings(new EditorKeybindingsManager(editorConfig)); - - return manager; - } - - /** - * Create in-memory. - */ - static inMemory(config: KeybindingsConfig = {}): KeybindingsManager { - return new KeybindingsManager(config); - } - - static async #loadFromFile(path: string): Promise { - try { - return await Bun.file(path).json(); - } catch (error) { - if (isEnoent(error)) return {}; - logger.warn("Failed to parse keybindings config", { path, error: String(error) }); - return {}; - } - } - - #buildMaps(): void { - this.#appActionToKeys.clear(); - - // Set defaults for app actions - for (const [action, keys] of Object.entries(DEFAULT_APP_KEYBINDINGS)) { - const keyArray = Array.isArray(keys) ? keys : [keys]; - this.#appActionToKeys.set( - action as AppAction, - keyArray.map(key => normalizeKeyId(key as KeyId)), - ); - } - - // Override with user config (app actions only) - for (const [action, keys] of Object.entries(this.config)) { - if (keys === undefined || !isAppAction(action)) continue; - const keyArray = Array.isArray(keys) ? keys : [keys]; - this.#appActionToKeys.set( - action, - keyArray.map(key => normalizeKeyId(key as KeyId)), - ); - } - } - - /** - * Check if input matches an app action. - */ - matches(data: string, action: AppAction): boolean { - const keys = this.#appActionToKeys.get(action); - if (!keys) return false; - for (const key of keys) { - if (matchesKey(data, key)) return true; - } - return false; - } - - /** - * Get keys bound to an app action. - */ - getKeys(action: AppAction): KeyId[] { - return this.#appActionToKeys.get(action) ?? []; - } - - /** - * Get display string for an action. - */ - getDisplayString(action: AppAction): string { - return formatKeyHints(this.getKeys(action)); - } - - /** - * Get the full effective config. - */ - getEffectiveConfig(): Required { - const result = { ...DEFAULT_KEYBINDINGS }; - for (const [action, keys] of Object.entries(this.config)) { - if (keys !== undefined) { - (result as KeybindingsConfig)[action as KeyAction] = keys; - } - } - return result; - } -} - -// Re-export for convenience -export type { EditorAction, KeyId }; +export type { Keybinding, KeybindingsConfig, KeyId }; +export { migrateKeybindingsConfigFile }; diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 251dc2964..29a49a7d8 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -90,6 +90,7 @@ const OpenAICompatSchema = Type.Object({ thinkingFormat: Type.Optional( Type.Union([ Type.Literal("openai"), + Type.Literal("openrouter"), Type.Literal("zai"), Type.Literal("qwen"), Type.Literal("qwen-chat-template"), diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 157b8a118..82d20b3b5 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -141,6 +141,55 @@ function isAlias(id: string): boolean { return !datePattern.test(id); } +/** + * Find an exact model reference match. + * Supports either a bare model id or a canonical provider/modelId reference. + * When matching by bare id, ambiguous matches across providers are rejected. + */ +export function findExactModelReferenceMatch( + modelReference: string, + availableModels: Model[], +): Model | undefined { + const trimmedReference = modelReference.trim(); + if (!trimmedReference) { + return undefined; + } + + const normalizedReference = trimmedReference.toLowerCase(); + + const canonicalMatches = availableModels.filter( + model => `${model.provider}/${model.id}`.toLowerCase() === normalizedReference, + ); + if (canonicalMatches.length === 1) { + return canonicalMatches[0]; + } + if (canonicalMatches.length > 1) { + return undefined; + } + + const slashIndex = trimmedReference.indexOf("/"); + if (slashIndex !== -1) { + const provider = trimmedReference.substring(0, slashIndex).trim(); + const modelId = trimmedReference.substring(slashIndex + 1).trim(); + if (provider && modelId) { + const providerMatches = availableModels.filter( + model => + model.provider.toLowerCase() === provider.toLowerCase() && + model.id.toLowerCase() === modelId.toLowerCase(), + ); + if (providerMatches.length === 1) { + return providerMatches[0]; + } + if (providerMatches.length > 1) { + return undefined; + } + } + } + + const idMatches = availableModels.filter(model => model.id.toLowerCase() === normalizedReference); + return idMatches.length === 1 ? idMatches[0] : undefined; +} + /** * Try to match a pattern to a model from the available models list. * Returns the matched model or undefined if no match found. @@ -150,17 +199,17 @@ function tryMatchModel( availableModels: Model[], context: ModelPreferenceContext, ): Model | undefined { - // Check for provider/modelId format (provider is everything before the first /) + // Try exact reference match first (handles provider/modelId and bare id with ambiguity rejection) + const exactRefMatch = findExactModelReferenceMatch(modelPattern, availableModels); + if (exactRefMatch) { + return exactRefMatch; + } + + // Check for provider/modelId format — fuzzy match within provider const slashIndex = modelPattern.indexOf("/"); if (slashIndex !== -1) { const provider = modelPattern.substring(0, slashIndex); const modelId = modelPattern.substring(slashIndex + 1); - const providerMatch = availableModels.find( - m => m.provider.toLowerCase() === provider.toLowerCase() && m.id.toLowerCase() === modelId.toLowerCase(), - ); - if (providerMatch) { - return providerMatch; - } const providerModels = availableModels.filter(m => m.provider.toLowerCase() === provider.toLowerCase()); if (providerModels.length > 0) { @@ -187,10 +236,9 @@ function tryMatchModel( return scored[0]?.model; } } - // No exact provider/model match - fall through to other matching } - // Check for exact ID match (case-insensitive) + // Exact ID match (case-insensitive) — with ambiguity across providers handled by preference const exactMatches = availableModels.filter(m => m.id.toLowerCase() === modelPattern.toLowerCase()); if (exactMatches.length > 0) { return pickPreferredModel(exactMatches, context); diff --git a/packages/coding-agent/src/export/html/template.css b/packages/coding-agent/src/export/html/template.css index d1d40b63f..4a5287052 100644 --- a/packages/coding-agent/src/export/html/template.css +++ b/packages/coding-agent/src/export/html/template.css @@ -2,6 +2,10 @@ :root { --line-height: 18px; /* 12px font * 1.5 */ + --sidebar-width: 400px; + --sidebar-min-width: 240px; + --sidebar-max-width: 840px; + --sidebar-resizer-width: 6px; } body { @@ -12,6 +16,11 @@ background: var(--body-bg); } + body.sidebar-resizing { + cursor: col-resize; + user-select: none; + } + #app { display: flex; min-height: 100vh; @@ -19,7 +28,9 @@ /* Sidebar */ #sidebar { - width: 400px; + width: var(--sidebar-width); + min-width: var(--sidebar-width); + max-width: var(--sidebar-width); background: var(--container-bg); flex-shrink: 0; display: flex; @@ -203,8 +214,28 @@ flex-shrink: 0; } + #sidebar-resizer { + width: var(--sidebar-resizer-width); + flex-shrink: 0; + position: sticky; + top: 0; + height: 100vh; + cursor: col-resize; + touch-action: none; + background: transparent; + border-right: 1px solid transparent; + } + + #sidebar-resizer:hover, + body.sidebar-resizing #sidebar-resizer { + background: var(--selectedBg); + border-right-color: var(--dim); + } + /* Main content */ #content { + flex: 1; + min-width: 0; flex: 1; overflow-y: auto; padding: var(--line-height) calc(var(--line-height) * 2); @@ -841,17 +872,19 @@ @media (max-width: 900px) { #sidebar { position: fixed; - left: -400px; - width: 400px; + transform: translateX(-100%); + width: min(var(--sidebar-width), 100vw); + min-width: 0; + max-width: 100vw; top: 0; bottom: 0; height: 100vh; z-index: 99; - transition: left 0.3s; + transition: transform 0.3s; } #sidebar.open { - left: 0; + transform: translateX(0); } #sidebar-overlay.open { @@ -866,6 +899,10 @@ display: block; } + #sidebar-resizer { + display: none; + } + #content { padding: var(--line-height) 16px; } @@ -875,15 +912,8 @@ } } - @media (max-width: 500px) { - #sidebar { - width: 100vw; - left: -100vw; - } - } - @media print { - #sidebar, #sidebar-toggle { display: none !important; } + #sidebar, #sidebar-toggle, #sidebar-resizer { display: none !important; } body { background: white; color: black; } #content { max-width: none; } } diff --git a/packages/coding-agent/src/export/html/template.generated.ts b/packages/coding-agent/src/export/html/template.generated.ts index 159081fa8..8be39c480 100644 --- a/packages/coding-agent/src/export/html/template.generated.ts +++ b/packages/coding-agent/src/export/html/template.generated.ts @@ -1,2 +1,2 @@ // Auto-generated by scripts/generate-template.ts - DO NOT EDIT -export const TEMPLATE = "\n\n\n \n \n Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; +export const TEMPLATE = "\n\n\n \n \n Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; diff --git a/packages/coding-agent/src/export/html/template.html b/packages/coding-agent/src/export/html/template.html index 3afb4beb3..0330e1307 100644 --- a/packages/coding-agent/src/export/html/template.html +++ b/packages/coding-agent/src/export/html/template.html @@ -28,6 +28,7 @@
+
diff --git a/packages/coding-agent/src/export/html/template.js b/packages/coding-agent/src/export/html/template.js index 6c3991c9f..06a9a4406 100644 --- a/packages/coding-agent/src/export/html/template.js +++ b/packages/coding-agent/src/export/html/template.js @@ -1279,6 +1279,113 @@ const sidebar = document.getElementById('sidebar'); const overlay = document.getElementById('sidebar-overlay'); const hamburger = document.getElementById('hamburger'); + const sidebarResizer = document.getElementById('sidebar-resizer'); + const SIDEBAR_WIDTH_STORAGE_KEY = 'pi-share:v1:sidebar-width'; + const MIN_CONTENT_WIDTH = 320; + + function isMobileLayout() { + return window.matchMedia('(max-width: 900px)').matches; + } + + function getSidebarBounds() { + const rootStyles = getComputedStyle(document.documentElement); + const minWidth = parseFloat(rootStyles.getPropertyValue('--sidebar-min-width')) || 240; + const maxWidth = parseFloat(rootStyles.getPropertyValue('--sidebar-max-width')) || 720; + const viewportMaxWidth = window.innerWidth - MIN_CONTENT_WIDTH; + return { + minWidth, + maxWidth: Math.max(minWidth, Math.min(maxWidth, viewportMaxWidth)) + }; + } + + function clampSidebarWidth(width) { + const { minWidth, maxWidth } = getSidebarBounds(); + return Math.max(minWidth, Math.min(maxWidth, width)); + } + + function applySidebarWidth(width) { + document.documentElement.style.setProperty('--sidebar-width', `${Math.round(clampSidebarWidth(width))}px`); + } + + function loadSidebarWidth() { + try { + const raw = localStorage.getItem(SIDEBAR_WIDTH_STORAGE_KEY); + if (raw === null) return null; + const width = Number(raw); + return Number.isFinite(width) ? width : null; + } catch { + return null; + } + } + + function saveSidebarWidth(width) { + try { + localStorage.setItem(SIDEBAR_WIDTH_STORAGE_KEY, String(Math.round(clampSidebarWidth(width)))); + } catch { + // Ignore storage failures (e.g. private browsing restrictions) + } + } + + function setupSidebarResize() { + const savedWidth = loadSidebarWidth(); + if (savedWidth !== null) { + applySidebarWidth(savedWidth); + } + + if (!sidebarResizer) return; + + let cleanupDrag = null; + + const stopDrag = (pointerId) => { + if (cleanupDrag) { + cleanupDrag(pointerId); + cleanupDrag = null; + } + }; + + sidebarResizer.addEventListener('pointerdown', (e) => { + if (isMobileLayout()) return; + + e.preventDefault(); + const startX = e.clientX; + const startWidth = sidebar.getBoundingClientRect().width; + document.body.classList.add('sidebar-resizing'); + sidebarResizer.setPointerCapture?.(e.pointerId); + + const onPointerMove = (event) => { + applySidebarWidth(startWidth + (event.clientX - startX)); + }; + + cleanupDrag = (pointerIdToRelease) => { + document.body.classList.remove('sidebar-resizing'); + sidebarResizer.releasePointerCapture?.(pointerIdToRelease); + window.removeEventListener('pointermove', onPointerMove); + window.removeEventListener('pointerup', onPointerUp); + window.removeEventListener('pointercancel', onPointerCancel); + saveSidebarWidth(sidebar.getBoundingClientRect().width); + }; + + const onPointerUp = (event) => stopDrag(event.pointerId); + const onPointerCancel = (event) => stopDrag(event.pointerId); + + window.addEventListener('pointermove', onPointerMove); + window.addEventListener('pointerup', onPointerUp); + window.addEventListener('pointercancel', onPointerCancel); + }); + + sidebarResizer.addEventListener('dblclick', () => { + if (isMobileLayout()) return; + applySidebarWidth(400); + saveSidebarWidth(400); + }); + + window.addEventListener('resize', () => { + if (isMobileLayout()) return; + applySidebarWidth(sidebar.getBoundingClientRect().width); + }); + } + + setupSidebarResize(); hamburger.addEventListener('click', () => { sidebar.classList.add('open'); diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 035597f64..f670e147c 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -56,7 +56,7 @@ import type { TodoItem } from "../../tools/todo-write"; import type { EventBus } from "../../utils/event-bus"; import type { SlashCommandInfo } from "../slash-commands"; -export type { AppAction, KeybindingsManager } from "../../config/keybindings"; +export type { AppKeybinding, KeybindingsManager } from "../../config/keybindings"; export type { ExecOptions, ExecResult } from "../../exec/exec"; export type { AgentToolResult, AgentToolUpdateCallback }; diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 54259be9b..4b6090602 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -15,6 +15,7 @@ import { $env, getProjectDir, logger, postmortem, setProjectDir, VERSION } from import chalk from "chalk"; import type { Args } from "./cli/args"; import { processFileArguments } from "./cli/file-processor"; +import { buildInitialMessage } from "./cli/initial-message"; import { listModels } from "./cli/list-models"; import { selectSession } from "./cli/session-picker"; import { findConfigFile } from "./config"; @@ -137,7 +138,7 @@ async function runInteractiveMode( } } - if (initialMessage) { + if (initialMessage !== undefined) { try { await session.prompt(initialMessage, { images: initialImages }); } catch (error: unknown) { @@ -161,33 +162,6 @@ async function runInteractiveMode( } } -async function prepareInitialMessage( - parsed: Args, - autoResizeImages: boolean, -): Promise<{ - initialMessage?: string; - initialImages?: ImageContent[]; -}> { - if (parsed.fileArgs.length === 0) { - return {}; - } - - const { text, images } = await processFileArguments(parsed.fileArgs, { autoResizeImages }); - - let initialMessage: string; - if (parsed.messages.length > 0) { - initialMessage = text + parsed.messages[0]; - parsed.messages.shift(); - } else { - initialMessage = text; - } - - return { - initialMessage, - initialImages: images.length > 0 ? images : undefined, - }; -} - function normalizePathForComparison(value: string): string { const resolved = path.resolve(value); let realPath = resolved; @@ -237,6 +211,21 @@ async function getChangelogForDisplay(parsed: Args): Promise } async function createSessionManager(parsed: Args, cwd: string): Promise { + if (parsed.fork) { + if (parsed.noSession) { + throw new Error("--fork requires session persistence"); + } + const forkSource = parsed.fork; + if (forkSource.includes("/") || forkSource.includes("\\") || forkSource.endsWith(".jsonl")) { + return await SessionManager.forkFrom(forkSource, cwd, parsed.sessionDir); + } + const match = await resolveResumableSession(forkSource, cwd, parsed.sessionDir); + if (!match) { + throw new Error(`Session "${forkSource}" not found.`); + } + return await SessionManager.forkFrom(match.session.path, cwd, parsed.sessionDir); + } + if (parsed.noSession) { return SessionManager.inMemory(); } @@ -565,22 +554,27 @@ export async function runRootCommand(parsed: Args, rawArgs: string[]): Promise { + const { pipedInput, fileText, fileImages } = await logger.timeAsync("prepareInitialMessage", async () => { const pipedInput = await readPipedInput(); - let { initialMessage, initialImages } = await prepareInitialMessage( - parsedArgs, - settings.get("images.autoResize"), - ); - if (pipedInput) { - initialMessage = initialMessage ? `${initialMessage}\n${pipedInput}` : pipedInput; + if (parsedArgs.fileArgs.length === 0) { + return { pipedInput }; } - return { pipedInput, initialMessage, initialImages }; + + const { text, images } = await processFileArguments(parsedArgs.fileArgs, { + autoResizeImages: settings.get("images.autoResize"), + }); + return { + pipedInput, + fileText: text, + fileImages: images, + }; + }); + const { initialMessage, initialImages } = buildInitialMessage({ + parsed: parsedArgs, + fileText, + fileImages, + stdinContent: pipedInput, }); - const initialMessage = initMsg; const autoPrint = pipedInput !== undefined && !parsedArgs.print && parsedArgs.mode === undefined; const isInteractive = !parsedArgs.print && !autoPrint && parsedArgs.mode === undefined; const mode = parsedArgs.mode || "text"; @@ -626,7 +620,7 @@ export async function runRootCommand(parsed: Args, rawArgs: string[]): Promise createSessionManager(parsedArgs, cwd)); // Handle --resume (no value): show session picker - if (parsedArgs.resume === true) { + if (parsedArgs.resume === true && !parsedArgs.fork) { const sessions = await logger.timeAsync("SessionManager.list", () => SessionManager.list(cwd, parsedArgs.sessionDir), ); diff --git a/packages/coding-agent/src/modes/components/custom-editor.ts b/packages/coding-agent/src/modes/components/custom-editor.ts index be956c20e..94ec30856 100644 --- a/packages/coding-agent/src/modes/components/custom-editor.ts +++ b/packages/coding-agent/src/modes/components/custom-editor.ts @@ -1,41 +1,41 @@ import { Editor, type KeyId, matchesKey, parseKittySequence } from "@oh-my-pi/pi-tui"; -import type { AppAction } from "../../config/keybindings"; +import type { AppKeybinding } from "../../config/keybindings"; type ConfigurableEditorAction = Extract< - AppAction, - | "interrupt" - | "clear" - | "exit" - | "suspend" - | "cycleThinkingLevel" - | "cycleModelForward" - | "cycleModelBackward" - | "selectModel" - | "expandTools" - | "toggleThinking" - | "externalEditor" - | "historySearch" - | "dequeue" - | "pasteImage" - | "copyPrompt" + AppKeybinding, + | "app.interrupt" + | "app.clear" + | "app.exit" + | "app.suspend" + | "app.thinking.cycle" + | "app.model.cycleForward" + | "app.model.cycleBackward" + | "app.model.select" + | "app.tools.expand" + | "app.thinking.toggle" + | "app.editor.external" + | "app.history.search" + | "app.message.dequeue" + | "app.clipboard.pasteImage" + | "app.clipboard.copyPrompt" >; const DEFAULT_ACTION_KEYS: Record = { - interrupt: ["escape"], - clear: ["ctrl+c"], - exit: ["ctrl+d"], - suspend: ["ctrl+z"], - cycleThinkingLevel: ["shift+tab"], - cycleModelForward: ["ctrl+p"], - cycleModelBackward: ["shift+ctrl+p"], - selectModel: ["ctrl+l"], - expandTools: ["ctrl+o"], - toggleThinking: ["ctrl+t"], - externalEditor: ["ctrl+g"], - historySearch: ["ctrl+r"], - dequeue: ["alt+up"], - pasteImage: ["ctrl+v"], - copyPrompt: ["alt+shift+c"], + "app.interrupt": ["escape"], + "app.clear": ["ctrl+c"], + "app.exit": ["ctrl+d"], + "app.suspend": ["ctrl+z"], + "app.thinking.cycle": ["shift+tab"], + "app.model.cycleForward": ["ctrl+p"], + "app.model.cycleBackward": ["shift+ctrl+p"], + "app.model.select": ["ctrl+l"], + "app.tools.expand": ["ctrl+o"], + "app.thinking.toggle": ["ctrl+t"], + "app.editor.external": ["ctrl+g"], + "app.history.search": ["ctrl+r"], + "app.message.dequeue": ["alt+up"], + "app.clipboard.pasteImage": ["ctrl+v"], + "app.clipboard.copyPrompt": ["alt+shift+c"], }; /** @@ -115,13 +115,13 @@ export class CustomEditor extends Editor { } // Intercept configured image paste (async - fires and handles result) - if (this.#matchesAction(data, "pasteImage") && this.onPasteImage) { + if (this.#matchesAction(data, "app.clipboard.pasteImage") && this.onPasteImage) { void this.onPasteImage(); return; } // Intercept configured external editor shortcut - if (this.#matchesAction(data, "externalEditor") && this.onExternalEditor) { + if (this.#matchesAction(data, "app.editor.external") && this.onExternalEditor) { this.onExternalEditor(); return; } @@ -133,56 +133,56 @@ export class CustomEditor extends Editor { } // Intercept configured suspend shortcut - if (this.#matchesAction(data, "suspend") && this.onSuspend) { + if (this.#matchesAction(data, "app.suspend") && this.onSuspend) { this.onSuspend(); return; } // Intercept configured thinking block visibility toggle - if (this.#matchesAction(data, "toggleThinking") && this.onToggleThinking) { + if (this.#matchesAction(data, "app.thinking.toggle") && this.onToggleThinking) { this.onToggleThinking(); return; } // Intercept configured model selector shortcut - if (this.#matchesAction(data, "selectModel") && this.onSelectModel) { + if (this.#matchesAction(data, "app.model.select") && this.onSelectModel) { this.onSelectModel(); return; } // Intercept configured history search shortcut - if (this.#matchesAction(data, "historySearch") && this.onHistorySearch) { + if (this.#matchesAction(data, "app.history.search") && this.onHistorySearch) { this.onHistorySearch(); return; } // Intercept configured tool output expansion shortcut - if (this.#matchesAction(data, "expandTools") && this.onExpandTools) { + if (this.#matchesAction(data, "app.tools.expand") && this.onExpandTools) { this.onExpandTools(); return; } // Intercept configured backward model cycling (check before forward cycling) - if (this.#matchesAction(data, "cycleModelBackward") && this.onCycleModelBackward) { + if (this.#matchesAction(data, "app.model.cycleBackward") && this.onCycleModelBackward) { this.onCycleModelBackward(); return; } // Intercept configured forward model cycling - if (this.#matchesAction(data, "cycleModelForward") && this.onCycleModelForward) { + if (this.#matchesAction(data, "app.model.cycleForward") && this.onCycleModelForward) { this.onCycleModelForward(); return; } // Intercept configured thinking level cycling - if (this.#matchesAction(data, "cycleThinkingLevel") && this.onCycleThinkingLevel) { + if (this.#matchesAction(data, "app.thinking.cycle") && this.onCycleThinkingLevel) { this.onCycleThinkingLevel(); return; } // Intercept configured interrupt shortcut. // Default behavior keeps autocomplete dismissal, but parent can prioritize global interrupt handling. - if (this.#matchesAction(data, "interrupt") && this.onEscape) { + if (this.#matchesAction(data, "app.interrupt") && this.onEscape) { if (!this.isShowingAutocomplete() || this.shouldBypassAutocompleteOnEscape?.()) { this.onEscape(); return; @@ -190,13 +190,13 @@ export class CustomEditor extends Editor { } // Intercept configured clear shortcut - if (this.#matchesAction(data, "clear") && this.onClear) { + if (this.#matchesAction(data, "app.clear") && this.onClear) { this.onClear(); return; } // Intercept configured exit shortcut (only when editor is empty) - if (this.#matchesAction(data, "exit")) { + if (this.#matchesAction(data, "app.exit")) { if (this.getText().length === 0 && this.onExit) { this.onExit(); } @@ -205,13 +205,13 @@ export class CustomEditor extends Editor { } // Intercept configured dequeue shortcut (restore queued message to editor) - if (this.#matchesAction(data, "dequeue") && this.onDequeue) { + if (this.#matchesAction(data, "app.message.dequeue") && this.onDequeue) { this.onDequeue(); return; } // Intercept configured copy-prompt shortcut - if (this.#matchesAction(data, "copyPrompt") && this.onCopyPrompt) { + if (this.#matchesAction(data, "app.clipboard.copyPrompt") && this.onCopyPrompt) { this.onCopyPrompt(); return; } diff --git a/packages/coding-agent/src/modes/components/keybinding-hints.ts b/packages/coding-agent/src/modes/components/keybinding-hints.ts index da567878a..9b1bf14b1 100644 --- a/packages/coding-agent/src/modes/components/keybinding-hints.ts +++ b/packages/coding-agent/src/modes/components/keybinding-hints.ts @@ -1,8 +1,8 @@ /** * Utilities for formatting keybinding hints in the UI. */ -import { type EditorAction, getEditorKeybindings, type KeyId } from "@oh-my-pi/pi-tui"; -import type { AppAction, KeybindingsManager } from "../../config/keybindings"; +import { getKeybindings, type Keybinding, type KeyId } from "@oh-my-pi/pi-tui"; +import type { AppKeybinding, KeybindingsManager } from "../../config/keybindings"; import { theme } from "../../modes/theme/theme"; /** @@ -17,14 +17,14 @@ function formatKeys(keys: KeyId[]): string { /** * Get display string for an editor action. */ -export function editorKey(action: EditorAction): string { - return formatKeys(getEditorKeybindings().getKeys(action)); +export function editorKey(action: Keybinding): string { + return formatKeys(getKeybindings().getKeys(action)); } /** * Get display string for an app action. */ -export function appKey(keybindings: KeybindingsManager, action: AppAction): string { +export function appKey(keybindings: KeybindingsManager, action: AppKeybinding): string { return formatKeys(keybindings.getKeys(action)); } @@ -32,11 +32,11 @@ export function appKey(keybindings: KeybindingsManager, action: AppAction): stri * Format a keybinding hint with consistent styling: dim key, muted description. * Looks up the key from editor keybindings automatically. * - * @param action - Editor action name (e.g., "selectConfirm", "expandTools") + * @param action - Keybinding action name (e.g., "tui.select.confirm", "app.tools.expand") * @param description - Description text (e.g., "to expand", "cancel") * @returns Formatted string with dim key and muted description */ -export function keyHint(action: EditorAction, description: string): string { +export function keyHint(action: Keybinding, description: string): string { return theme.fg("dim", editorKey(action)) + theme.fg("muted", ` ${description}`); } @@ -45,11 +45,11 @@ export function keyHint(action: EditorAction, description: string): string { * Requires the KeybindingsManager instance. * * @param keybindings - KeybindingsManager instance - * @param action - App action name (e.g., "interrupt", "externalEditor") + * @param action - App keybinding name (e.g., "app.interrupt", "app.editor.external") * @param description - Description text * @returns Formatted string with dim key and muted description */ -export function appKeyHint(keybindings: KeybindingsManager, action: AppAction, description: string): string { +export function appKeyHint(keybindings: KeybindingsManager, action: AppKeybinding, description: string): string { return theme.fg("dim", appKey(keybindings, action)) + theme.fg("muted", ` ${description}`); } diff --git a/packages/coding-agent/src/modes/components/login-dialog.ts b/packages/coding-agent/src/modes/components/login-dialog.ts index e7a0e1cdc..d15295a70 100644 --- a/packages/coding-agent/src/modes/components/login-dialog.ts +++ b/packages/coding-agent/src/modes/components/login-dialog.ts @@ -1,5 +1,5 @@ import { getOAuthProviders } from "@oh-my-pi/pi-ai"; -import { Container, getEditorKeybindings, Input, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; +import { Container, getKeybindings, Input, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; import { theme } from "../../modes/theme/theme"; import { openPath } from "../../utils/open"; import { DynamicBorder } from "./dynamic-border"; @@ -151,9 +151,9 @@ export class LoginDialogComponent extends Container { } handleInput(data: string): void { - const kb = getEditorKeybindings(); + const kb = getKeybindings(); - if (kb.matches(data, "selectCancel")) { + if (kb.matches(data, "tui.select.cancel")) { this.#cancel(); return; } diff --git a/packages/coding-agent/src/modes/components/user-message.ts b/packages/coding-agent/src/modes/components/user-message.ts index 5b05e3789..5f1da9317 100644 --- a/packages/coding-agent/src/modes/components/user-message.ts +++ b/packages/coding-agent/src/modes/components/user-message.ts @@ -1,6 +1,11 @@ import { Container, Markdown, Spacer } from "@oh-my-pi/pi-tui"; import { getMarkdownTheme, theme } from "../../modes/theme/theme"; +// OSC 133 shell integration: marks prompt zones for terminal multiplexers +const OSC133_ZONE_START = "\x1b]133;A\x07"; +const OSC133_ZONE_END = "\x1b]133;B\x07"; +const OSC133_ZONE_FINAL = "\x1b]133;C\x07"; + /** * Component that renders a user message */ @@ -19,4 +24,15 @@ export class UserMessageComponent extends Container { }), ); } + + override render(width: number): string[] { + const lines = super.render(width); + if (lines.length === 0) { + return lines; + } + + lines[0] = OSC133_ZONE_START + lines[0]; + lines[lines.length - 1] = lines[lines.length - 1] + OSC133_ZONE_END + OSC133_ZONE_FINAL; + return lines; + } } diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index 82f7aa556..9af32c5d3 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -26,7 +26,7 @@ export class InputController { constructor(private ctx: InteractiveModeContext) {} setupKeyHandlers(): void { - this.ctx.editor.setActionKeys("interrupt", this.ctx.keybindings.getKeys("interrupt")); + this.ctx.editor.setActionKeys("app.interrupt", this.ctx.keybindings.getKeys("app.interrupt")); this.ctx.editor.shouldBypassAutocompleteOnEscape = () => Boolean( this.ctx.loadingAnimation || @@ -83,68 +83,74 @@ export class InputController { } }; - this.ctx.editor.setActionKeys("clear", this.ctx.keybindings.getKeys("clear")); + this.ctx.editor.setActionKeys("app.clear", this.ctx.keybindings.getKeys("app.clear")); this.ctx.editor.onClear = () => this.handleCtrlC(); - this.ctx.editor.setActionKeys("exit", this.ctx.keybindings.getKeys("exit")); + this.ctx.editor.setActionKeys("app.exit", this.ctx.keybindings.getKeys("app.exit")); this.ctx.editor.onExit = () => this.handleCtrlD(); - this.ctx.editor.setActionKeys("suspend", this.ctx.keybindings.getKeys("suspend")); + this.ctx.editor.setActionKeys("app.suspend", this.ctx.keybindings.getKeys("app.suspend")); this.ctx.editor.onSuspend = () => this.handleCtrlZ(); - this.ctx.editor.setActionKeys("cycleThinkingLevel", this.ctx.keybindings.getKeys("cycleThinkingLevel")); + this.ctx.editor.setActionKeys("app.thinking.cycle", this.ctx.keybindings.getKeys("app.thinking.cycle")); this.ctx.editor.onCycleThinkingLevel = () => this.cycleThinkingLevel(); - this.ctx.editor.setActionKeys("cycleModelForward", this.ctx.keybindings.getKeys("cycleModelForward")); + this.ctx.editor.setActionKeys("app.model.cycleForward", this.ctx.keybindings.getKeys("app.model.cycleForward")); this.ctx.editor.onCycleModelForward = () => this.cycleRoleModel(); - this.ctx.editor.setActionKeys("cycleModelBackward", this.ctx.keybindings.getKeys("cycleModelBackward")); + this.ctx.editor.setActionKeys("app.model.cycleBackward", this.ctx.keybindings.getKeys("app.model.cycleBackward")); this.ctx.editor.onCycleModelBackward = () => this.cycleRoleModel({ temporary: true }); this.ctx.editor.onQuickSelectModel = () => this.ctx.showModelSelector({ temporaryOnly: true }); // Global debug handler on TUI (works regardless of focus) this.ctx.ui.onDebug = () => this.ctx.showDebugSelector(); - this.ctx.editor.setActionKeys("selectModel", this.ctx.keybindings.getKeys("selectModel")); + this.ctx.editor.setActionKeys("app.model.select", this.ctx.keybindings.getKeys("app.model.select")); this.ctx.editor.onSelectModel = () => this.ctx.showModelSelector(); - this.ctx.editor.setActionKeys("historySearch", this.ctx.keybindings.getKeys("historySearch")); + this.ctx.editor.setActionKeys("app.history.search", this.ctx.keybindings.getKeys("app.history.search")); this.ctx.editor.onHistorySearch = () => this.ctx.showHistorySearch(); - this.ctx.editor.setActionKeys("toggleThinking", this.ctx.keybindings.getKeys("toggleThinking")); + this.ctx.editor.setActionKeys("app.thinking.toggle", this.ctx.keybindings.getKeys("app.thinking.toggle")); this.ctx.editor.onToggleThinking = () => this.ctx.toggleThinkingBlockVisibility(); - this.ctx.editor.setActionKeys("externalEditor", this.ctx.keybindings.getKeys("externalEditor")); + this.ctx.editor.setActionKeys("app.editor.external", this.ctx.keybindings.getKeys("app.editor.external")); this.ctx.editor.onExternalEditor = () => void this.openExternalEditor(); this.ctx.editor.onShowHotkeys = () => this.ctx.handleHotkeysCommand(); - this.ctx.editor.setActionKeys("pasteImage", this.ctx.keybindings.getKeys("pasteImage")); + this.ctx.editor.setActionKeys( + "app.clipboard.pasteImage", + this.ctx.keybindings.getKeys("app.clipboard.pasteImage"), + ); this.ctx.editor.onPasteImage = () => this.handleImagePaste(); - this.ctx.editor.setActionKeys("copyPrompt", this.ctx.keybindings.getKeys("copyPrompt")); + this.ctx.editor.setActionKeys( + "app.clipboard.copyPrompt", + this.ctx.keybindings.getKeys("app.clipboard.copyPrompt"), + ); this.ctx.editor.onCopyPrompt = () => this.handleCopyPrompt(); - this.ctx.editor.setActionKeys("expandTools", this.ctx.keybindings.getKeys("expandTools")); + this.ctx.editor.setActionKeys("app.tools.expand", this.ctx.keybindings.getKeys("app.tools.expand")); this.ctx.editor.onExpandTools = () => this.toggleToolOutputExpansion(); - this.ctx.editor.setActionKeys("dequeue", this.ctx.keybindings.getKeys("dequeue")); + this.ctx.editor.setActionKeys("app.message.dequeue", this.ctx.keybindings.getKeys("app.message.dequeue")); this.ctx.editor.onDequeue = () => this.handleDequeue(); this.ctx.editor.clearCustomKeyHandlers(); // Wire up extension shortcuts this.registerExtensionShortcuts(); - const planModeKeys = this.ctx.keybindings.getKeys("togglePlanMode"); + const planModeKeys = this.ctx.keybindings.getKeys("app.plan.toggle"); for (const key of planModeKeys) { this.ctx.editor.setCustomKeyHandler(key, () => void this.ctx.handlePlanModeCommand()); } - for (const key of this.ctx.keybindings.getKeys("newSession")) { + for (const key of this.ctx.keybindings.getKeys("app.session.new")) { this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.handleClearCommand()); } - for (const key of this.ctx.keybindings.getKeys("tree")) { + for (const key of this.ctx.keybindings.getKeys("app.session.tree")) { this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.showTreeSelector()); } - for (const key of this.ctx.keybindings.getKeys("fork")) { + for (const key of this.ctx.keybindings.getKeys("app.session.fork")) { this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.showUserMessageSelector()); } - for (const key of this.ctx.keybindings.getKeys("resume")) { + for (const key of this.ctx.keybindings.getKeys("app.session.resume")) { this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.showSessionSelector()); } - for (const key of this.ctx.keybindings.getKeys("followUp")) { + for (const key of this.ctx.keybindings.getKeys("app.message.followUp")) { this.ctx.editor.setCustomKeyHandler(key, () => void this.handleFollowUp()); } - for (const key of this.ctx.keybindings.getKeys("toggleSTT")) { + for (const key of this.ctx.keybindings.getKeys("app.stt.toggle")) { this.ctx.editor.setCustomKeyHandler(key, () => void this.ctx.handleSTTToggle()); } - for (const key of this.ctx.keybindings.getKeys("copyLine")) { + for (const key of this.ctx.keybindings.getKeys("app.clipboard.copyLine")) { this.ctx.editor.setCustomKeyHandler(key, () => this.handleCopyCurrentLine()); } diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index ab0c53494..841a2654a 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -273,7 +273,7 @@ export class InteractiveMode implements InteractiveModeContext { async init(): Promise { if (this.isInitialized) return; - this.keybindings = await logger.timeAsync("InteractiveMode.init:keybindings", () => KeybindingsManager.create()); + this.keybindings = logger.time("InteractiveMode.init:keybindings", () => KeybindingsManager.create()); // Register session manager flush for signal handlers (SIGINT, SIGTERM, SIGHUP) this.#cleanupUnsubscribe = postmortem.register("session-manager-flush", () => this.sessionManager.flush()); diff --git a/packages/coding-agent/src/modes/print-mode.ts b/packages/coding-agent/src/modes/print-mode.ts index 5340ccead..a588713b6 100644 --- a/packages/coding-agent/src/modes/print-mode.ts +++ b/packages/coding-agent/src/modes/print-mode.ts @@ -146,7 +146,7 @@ export async function runPrintMode(session: AgentSession, options: PrintModeOpti }); // Send initial message with attachments - if (initialMessage) { + if (initialMessage !== undefined) { await session.prompt(initialMessage, { images: initialImages }); } diff --git a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts index d991b8b03..f5d20b7df 100644 --- a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts +++ b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts @@ -2,7 +2,7 @@ import { type AutocompleteItem, type AutocompleteProvider, CombinedAutocompleteProvider, - getEditorKeybindings, + getKeybindings, type SlashCommand, } from "@oh-my-pi/pi-tui"; import { formatKeyHints, type KeybindingsManager } from "../config/keybindings"; @@ -174,26 +174,26 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { export function createPromptActionAutocompleteProvider( options: PromptActionAutocompleteOptions, ): PromptActionAutocompleteProvider { - const editorKeybindings = getEditorKeybindings(); + const editorKeybindings = getKeybindings(); const actions: PromptActionDefinition[] = [ { id: "copy-line", label: "Copy current line", - description: formatKeyHints(options.keybindings.getKeys("copyLine")), + description: formatKeyHints(options.keybindings.getKeys("app.clipboard.copyLine")), keywords: ["copy", "line", "clipboard", "current"], execute: options.copyCurrentLine, }, { id: "copy-prompt", label: "Copy whole prompt", - description: formatKeyHints(options.keybindings.getKeys("copyPrompt")), + description: formatKeyHints(options.keybindings.getKeys("app.clipboard.copyPrompt")), keywords: ["copy", "prompt", "clipboard", "message"], execute: options.copyPrompt, }, { id: "undo", label: "Undo", - description: formatKeyHints(editorKeybindings.getKeys("undo")), + description: formatKeyHints(editorKeybindings.getKeys("tui.editor.undo")), keywords: ["undo", "revert", "edit", "history"], execute: options.undo, }, @@ -214,14 +214,14 @@ export function createPromptActionAutocompleteProvider( { id: "cursor-line-start", label: "Move cursor to beginning of line", - description: formatKeyHints(editorKeybindings.getKeys("cursorLineStart")), + description: formatKeyHints(editorKeybindings.getKeys("tui.editor.cursorLineStart")), keywords: ["move", "cursor", "line", "start", "beginning", "home"], execute: options.moveCursorToLineStart, }, { id: "cursor-line-end", label: "Move cursor to end of line", - description: formatKeyHints(editorKeybindings.getKeys("cursorLineEnd")), + description: formatKeyHints(editorKeybindings.getKeys("tui.editor.cursorLineEnd")), keywords: ["move", "cursor", "line", "end"], execute: options.moveCursorToLineEnd, }, diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index 9f80dce91..eef4aa9f3 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -1679,6 +1679,7 @@ export function getCurrentThemeName(): string | undefined { var currentSymbolPresetOverride: SymbolPreset | undefined; var currentColorBlindMode: boolean = false; var themeWatcher: fs.FSWatcher | undefined; +var themeReloadTimer: NodeJS.Timeout | undefined; var sigwinchHandler: (() => void) | undefined; var autoDetectedTheme: boolean = false; var autoDarkTheme: string = "dark"; @@ -1888,11 +1889,7 @@ export function isValidSymbolPreset(preset: string): preset is SymbolPreset { } async function startThemeWatcher(): Promise { - // Stop existing watcher if any - if (themeWatcher) { - themeWatcher.close(); - themeWatcher = undefined; - } + stopThemeWatcher(); // Only watch if it's a custom theme (not built-in) if (!currentThemeName || currentThemeName === "dark" || currentThemeName === "light") { @@ -1900,54 +1897,62 @@ async function startThemeWatcher(): Promise { } const customThemesDir = getCustomThemesDir(); - const themeFile = path.join(customThemesDir, `${currentThemeName}.json`); + const watchedThemeName = currentThemeName; + const watchedFileName = `${watchedThemeName}.json`; + const themeFile = path.join(customThemesDir, watchedFileName); // Only watch if the file exists if (!fs.existsSync(themeFile)) { return; } - try { - themeWatcher = fs.watch(themeFile, eventType => { - if (eventType === "change") { - // Debounce rapid changes - setTimeout(() => { - loadTheme(currentThemeName!, getCurrentThemeOptions()) - .then(loadedTheme => { - theme = loadedTheme; - if (onThemeChangeCallback) { - onThemeChangeCallback(); - } - }) - .catch(err => { - logger.debug("Theme reload error during file change", { error: String(err) }); - }); - }, 100); - } else if (eventType === "rename") { - // File was deleted or renamed - fall back to default theme - setTimeout(() => { - if (!fs.existsSync(themeFile)) { - currentThemeName = "dark"; - loadTheme("dark", getCurrentThemeOptions()) - .then(loadedTheme => { - theme = loadedTheme; - if (onThemeChangeCallback) { - onThemeChangeCallback(); - } - }) - .catch(err => { - logger.debug("Theme reload error during rename fallback", { error: String(err) }); - }); - if (themeWatcher) { - themeWatcher.close(); - themeWatcher = undefined; - } - } - }, 100); + const scheduleReload = () => { + if (themeReloadTimer) { + clearTimeout(themeReloadTimer); + } + themeReloadTimer = setTimeout(() => { + themeReloadTimer = undefined; + + // Ignore stale timers after switching themes or stopping the watcher + if (currentThemeName !== watchedThemeName) { + return; } + + // Keep the last successfully loaded theme active if the file is temporarily missing + if (!fs.existsSync(themeFile)) { + return; + } + + loadTheme(watchedThemeName, getCurrentThemeOptions()) + .then(loadedTheme => { + theme = loadedTheme; + if (onThemeChangeCallback) { + onThemeChangeCallback(); + } + }) + .catch(() => { + // Ignore errors (file might be in invalid state while being edited) + }); + }, 100); + }; + + try { + themeWatcher = fs.watch(customThemesDir, (_eventType, filename) => { + if (currentThemeName !== watchedThemeName) { + return; + } + if (!filename) { + scheduleReload(); + return; + } + const changedFile = String(filename); + if (changedFile !== watchedFileName) { + return; + } + scheduleReload(); }); - } catch (err) { - logger.debug("Failed to start theme watcher", { error: String(err) }); + } catch { + // Ignore errors starting watcher } } @@ -2023,6 +2028,10 @@ function stopSigwinchListener(): void { } export function stopThemeWatcher(): void { + if (themeReloadTimer) { + clearTimeout(themeReloadTimer); + themeReloadTimer = undefined; + } if (themeWatcher) { themeWatcher.close(); themeWatcher = undefined; diff --git a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts index f8415d443..3f1b75f49 100644 --- a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts +++ b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts @@ -1,10 +1,10 @@ -import type { AppAction, KeybindingsManager } from "../../config/keybindings"; +import type { AppKeybinding, KeybindingsManager } from "../../config/keybindings"; export interface HotkeysMarkdownBindings { keybindings: Pick; } -function appKey(bindings: HotkeysMarkdownBindings, action: AppAction): string { +function appKey(bindings: HotkeysMarkdownBindings, action: AppKeybinding): string { return bindings.keybindings.getDisplayString(action) || "Disabled"; } @@ -26,29 +26,29 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string "| `Ctrl+W` / `Option+Backspace` | Delete word backwards |", "| `Ctrl+U` | Delete to start of line |", "| `Ctrl+K` | Delete to end of line |", - `| \`${appKey(bindings, "copyLine")}\` | Copy current line |`, - `| \`${appKey(bindings, "copyPrompt")}\` | Copy whole prompt |`, + `| \`${appKey(bindings, "app.clipboard.copyLine")}\` | Copy current line |`, + `| \`${appKey(bindings, "app.clipboard.copyPrompt")}\` | Copy whole prompt |`, "", "**Other**", "| Key | Action |", "|-----|--------|", "| `Tab` | Path completion / accept autocomplete |", - `| \`${appKey(bindings, "interrupt")}\` | Cancel autocomplete / interrupt active work |`, - `| \`${appKey(bindings, "clear")}\` | Clear editor (first) / exit (second) |`, - `| \`${appKey(bindings, "exit")}\` | Exit (when editor is empty) |`, - `| \`${appKey(bindings, "suspend")}\` | Suspend to background |`, - `| \`${appKey(bindings, "cycleThinkingLevel")}\` | Cycle thinking level |`, - `| \`${appKey(bindings, "cycleModelForward")}\` | Cycle role models (slow/default/smol) |`, - `| \`${appKey(bindings, "cycleModelBackward")}\` | Cycle role models (temporary) |`, + `| \`${appKey(bindings, "app.interrupt")}\` | Cancel autocomplete / interrupt active work |`, + `| \`${appKey(bindings, "app.clear")}\` | Clear editor (first) / exit (second) |`, + `| \`${appKey(bindings, "app.exit")}\` | Exit (when editor is empty) |`, + `| \`${appKey(bindings, "app.suspend")}\` | Suspend to background |`, + `| \`${appKey(bindings, "app.thinking.cycle")}\` | Cycle thinking level |`, + `| \`${appKey(bindings, "app.model.cycleForward")}\` | Cycle role models (slow/default/smol) |`, + `| \`${appKey(bindings, "app.model.cycleBackward")}\` | Cycle role models (temporary) |`, "| `Alt+P` | Select model (temporary) |", - `| \`${appKey(bindings, "selectModel")}\` | Select model (set roles) |`, - `| \`${appKey(bindings, "togglePlanMode")}\` | Toggle plan mode |`, - `| \`${appKey(bindings, "historySearch")}\` | Search prompt history |`, - `| \`${appKey(bindings, "expandTools")}\` | Toggle tool output expansion |`, - `| \`${appKey(bindings, "toggleThinking")}\` | Toggle thinking block visibility |`, - `| \`${appKey(bindings, "externalEditor")}\` | Edit message in external editor |`, - `| \`${appKey(bindings, "pasteImage")}\` | Paste image from clipboard |`, - `| \`${appKey(bindings, "toggleSTT")}\` | Toggle speech-to-text recording |`, + `| \`${appKey(bindings, "app.model.select")}\` | Select model (set roles) |`, + `| \`${appKey(bindings, "app.plan.toggle")}\` | Toggle plan mode |`, + `| \`${appKey(bindings, "app.history.search")}\` | Search prompt history |`, + `| \`${appKey(bindings, "app.tools.expand")}\` | Toggle tool output expansion |`, + `| \`${appKey(bindings, "app.thinking.toggle")}\` | Toggle thinking block visibility |`, + `| \`${appKey(bindings, "app.editor.external")}\` | Edit message in external editor |`, + `| \`${appKey(bindings, "app.clipboard.pasteImage")}\` | Paste image from clipboard |`, + `| \`${appKey(bindings, "app.stt.toggle")}\` | Toggle speech-to-text recording |`, "| `#` | Open prompt actions |", "| `/` | Slash commands |", "| `!` | Run bash command |", diff --git a/packages/coding-agent/src/modes/utils/ui-helpers.ts b/packages/coding-agent/src/modes/utils/ui-helpers.ts index f3dd0642e..3632053bb 100644 --- a/packages/coding-agent/src/modes/utils/ui-helpers.ts +++ b/packages/coding-agent/src/modes/utils/ui-helpers.ts @@ -474,7 +474,7 @@ export class UiHelpers { const queuedText = theme.fg("dim", `${entry.label}: ${entry.message}`); this.ctx.pendingMessagesContainer.addChild(new TruncatedText(queuedText, 1, 0)); } - const dequeueKey = this.ctx.keybindings.getDisplayString("dequeue") || "Alt+Up"; + const dequeueKey = this.ctx.keybindings.getDisplayString("app.message.dequeue") || "Alt+Up"; const hintText = theme.fg("dim", `${theme.tree.hook} ${dequeueKey} to edit`); this.ctx.pendingMessagesContainer.addChild(new TruncatedText(hintText, 1, 0)); } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 6642d720a..108c777c7 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -782,7 +782,6 @@ export class AgentSession { attempt: this.#retryAttempt, }); this.#retryAttempt = 0; - this.#resolveRetry(); } } @@ -858,6 +857,7 @@ export class AgentSession { const didRetry = await this.#handleRetryableError(msg); if (didRetry) return; // Retry was initiated, don't proceed to compaction } + this.#resolveRetry(); if (msg.stopReason === "aborted" && this.#checkpointState) { this.#checkpointState = undefined; @@ -4495,7 +4495,9 @@ export class AgentSession { } #isTransientErrorMessage(errorMessage: string): boolean { - return /overloaded|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server error|internal error|connection.?error|unable to connect|fetch failed|retry delay|stream stall/i.test( + // Match: overloaded_error, provider returned error, rate limit, 429, 500, 502, 503, 504, + // service unavailable, network/connection errors, fetch failed, terminated, retry delay exceeded + return /overloaded|provider.?returned.?error|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server.?error|internal.?error|network.?error|connection.?error|connection.?refused|other side closed|fetch failed|upstream.?connect|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall/i.test( errorMessage, ); } diff --git a/packages/coding-agent/src/utils/child-process.ts b/packages/coding-agent/src/utils/child-process.ts new file mode 100644 index 000000000..8610ab171 --- /dev/null +++ b/packages/coding-agent/src/utils/child-process.ts @@ -0,0 +1,88 @@ +import type { ChildProcess } from "node:child_process"; + +const EXIT_STDIO_GRACE_MS = 100; + +/** + * Wait for a child process to terminate without hanging on inherited stdio handles. + * + * Daemonized descendants can inherit the child's stdout/stderr pipe handles. In that + * case the child emits `exit`, but `close` can hang forever even though the original + * process is already gone. We wait briefly for stdio to end, then forcibly stop + * tracking the inherited handles. + */ +export function waitForChildProcess(child: ChildProcess): Promise { + const { promise, resolve, reject } = Promise.withResolvers(); + + let settled = false; + let exited = false; + let exitCode: number | null = null; + let postExitTimer: NodeJS.Timeout | undefined; + let stdoutEnded = child.stdout === null; + let stderrEnded = child.stderr === null; + + const cleanup = () => { + if (postExitTimer) { + clearTimeout(postExitTimer); + postExitTimer = undefined; + } + child.removeListener("error", onError); + child.removeListener("exit", onExit); + child.removeListener("close", onClose); + child.stdout?.removeListener("end", onStdoutEnd); + child.stderr?.removeListener("end", onStderrEnd); + }; + + const finalize = (code: number | null) => { + if (settled) return; + settled = true; + cleanup(); + child.stdout?.destroy(); + child.stderr?.destroy(); + resolve(code); + }; + + const maybeFinalizeAfterExit = () => { + if (!exited || settled) return; + if (stdoutEnded && stderrEnded) { + finalize(exitCode); + } + }; + + const onStdoutEnd = () => { + stdoutEnded = true; + maybeFinalizeAfterExit(); + }; + + const onStderrEnd = () => { + stderrEnded = true; + maybeFinalizeAfterExit(); + }; + + const onError = (err: Error) => { + if (settled) return; + settled = true; + cleanup(); + reject(err); + }; + + const onExit = (code: number | null) => { + exited = true; + exitCode = code; + maybeFinalizeAfterExit(); + if (!settled) { + postExitTimer = setTimeout(() => finalize(code), EXIT_STDIO_GRACE_MS); + } + }; + + const onClose = (code: number | null) => { + finalize(code); + }; + + child.stdout?.once("end", onStdoutEnd); + child.stderr?.once("end", onStderrEnd); + child.once("error", onError); + child.once("exit", onExit); + child.once("close", onClose); + + return promise; +} diff --git a/packages/coding-agent/test/args.test.ts b/packages/coding-agent/test/args.test.ts index 8f439396a..0ebea8143 100644 --- a/packages/coding-agent/test/args.test.ts +++ b/packages/coding-agent/test/args.test.ts @@ -86,6 +86,13 @@ describe("parseArgs", () => { }); }); + describe("--fork flag", () => { + test("parses --fork with session ID", () => { + const result = parseArgs(["--fork", "abc123"]); + expect(result.fork).toBe("abc123"); + }); + }); + describe("flags with values", () => { test("parses --provider", () => { const result = parseArgs(["--provider", "openai"]); diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 9c961c13b..1e6e61187 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -246,7 +246,7 @@ describe("validateLineRef", () => { describe("applyHashlineEdits — replace", () => { it("replaces single line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), lines: ["BBB"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(2, "bbb"), lines: ["BBB"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nBBB\nccc"); @@ -255,7 +255,9 @@ describe("applyHashlineEdits — replace", () => { it("range replace (shrink)", () => { const content = "aaa\nbbb\nccc\nddd"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["ONE"] }]; + const edits: HashlineEdit[] = [ + { op: "replace_range", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["ONE"] }, + ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nONE\nddd"); @@ -264,7 +266,7 @@ describe("applyHashlineEdits — replace", () => { it("range replace (same count)", () => { const content = "aaa\nbbb\nccc\nddd"; const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["XXX", "YYY"] }, + { op: "replace_range", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: ["XXX", "YYY"] }, ]; const result = applyHashlineEdits(content, edits); @@ -274,7 +276,7 @@ describe("applyHashlineEdits — replace", () => { it("replaces first line", () => { const content = "first\nsecond\nthird"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(1, "first"), lines: ["FIRST"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(1, "first"), lines: ["FIRST"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("FIRST\nsecond\nthird"); @@ -283,7 +285,7 @@ describe("applyHashlineEdits — replace", () => { it("replaces last line", () => { const content = "first\nsecond\nthird"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(3, "third"), lines: ["THIRD"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(3, "third"), lines: ["THIRD"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("first\nsecond\nTHIRD"); @@ -298,7 +300,7 @@ describe("applyHashlineEdits — replace", () => { describe("applyHashlineEdits — delete", () => { it("deletes single line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(2, "bbb"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nccc"); @@ -307,7 +309,9 @@ describe("applyHashlineEdits — delete", () => { it("deletes range of lines", () => { const content = "aaa\nbbb\nccc\nddd"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: [] }]; + const edits: HashlineEdit[] = [ + { op: "replace_range", pos: makeTag(2, "bbb"), end: makeTag(3, "ccc"), lines: [] }, + ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nddd"); @@ -315,7 +319,7 @@ describe("applyHashlineEdits — delete", () => { it("deletes first line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(1, "aaa"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(1, "aaa"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("bbb\nccc"); @@ -323,7 +327,7 @@ describe("applyHashlineEdits — delete", () => { it("deletes last line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(3, "ccc"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(3, "ccc"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb"); @@ -331,7 +335,7 @@ describe("applyHashlineEdits — delete", () => { it("replaces line with blank line when lines is ['']", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), lines: [""] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(2, "bbb"), lines: [""] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\n\nccc"); @@ -380,7 +384,7 @@ describe("applyHashlineEdits — append", () => { it("inserts at EOF without anchors", () => { const content = "aaa\nbbb"; - const edits = [{ op: "append", lines: ["NEW"] }] as unknown as HashlineEdit[]; + const edits: HashlineEdit[] = [{ op: "append_eof", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb\nNEW"); @@ -389,7 +393,7 @@ describe("applyHashlineEdits — append", () => { it("inserts at EOF into empty file without anchors", () => { const content = ""; - const edits = [{ op: "append", lines: ["NEW"] }] as unknown as HashlineEdit[]; + const edits: HashlineEdit[] = [{ op: "append_eof", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("NEW"); @@ -398,7 +402,7 @@ describe("applyHashlineEdits — append", () => { it("insert at EOF with empty dst inserts a trailing empty line", () => { const content = "aaa\nbbb"; - const edits = [{ op: "append", lines: [] }] as unknown as HashlineEdit[]; + const edits: HashlineEdit[] = [{ op: "append_eof", lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb\n"); @@ -435,7 +439,7 @@ describe("applyHashlineEdits — prepend", () => { it("prepends at BOF without anchor", () => { const content = "aaa\nbbb"; - const edits = [{ op: "prepend", lines: ["NEW"] }] as unknown as HashlineEdit[]; + const edits: HashlineEdit[] = [{ op: "prepend_bof", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("NEW\naaa\nbbb"); expect(result.firstChangedLine).toBe(1); @@ -463,7 +467,7 @@ describe("applyHashlineEdits — prepend", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [ { op: "prepend", pos: makeTag(2, "bbb"), lines: ["BEFORE"] }, - { op: "replace", pos: makeTag(2, "bbb"), lines: ["BBB"] }, + { op: "replace_line", pos: makeTag(2, "bbb"), lines: ["BBB"] }, ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nBEFORE\nBBB\nccc"); @@ -482,7 +486,7 @@ describe("applyHashlineEdits — heuristics", () => { const srcHash = computeLineHash(2, "bbb"); const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_line", pos: parseTag(`2#${srcHash}export function foo(a, b) {}`), // comma in trailing content lines: ["BBB"], }, @@ -496,7 +500,7 @@ describe("applyHashlineEdits — heuristics", () => { const content = ["import { foo } from 'x';", "import { bar } from 'y';", "const x = 1;"].join("\n"); const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_range", pos: makeTag(1, "import { foo } from 'x';"), end: makeTag(2, "import { bar } from 'y';"), lines: ["import {foo} from 'x';", "import { bar } from 'y';", "// added"], @@ -514,7 +518,7 @@ describe("applyHashlineEdits — heuristics", () => { it("treats same-line ranges as single-line replacements", () => { const content = "aaa\nbbb\nccc"; const good = makeTag(2, "bbb"); - const edits: HashlineEdit[] = [{ op: "replace", pos: good, end: good, lines: ["BBB"] }]; + const edits: HashlineEdit[] = [{ op: "replace_range", pos: good, end: good, lines: ["BBB"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nBBB\nccc"); }); @@ -523,7 +527,7 @@ describe("applyHashlineEdits — heuristics", () => { const content = "if (ok) {\n run();\n}\nafter();"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_range", pos: makeTag(1, "if (ok) {"), end: makeTag(2, " run();"), lines: ["if (ok) {", " runSafe();", "}"], @@ -538,7 +542,7 @@ describe("applyHashlineEdits — heuristics", () => { const content = "start\n oldCall();\nnextCall();\nafter();"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_range", pos: makeTag(1, "start"), end: makeTag(2, " oldCall();"), lines: ["start", " newCall();", "nextCall();"], @@ -553,7 +557,7 @@ describe("applyHashlineEdits — heuristics", () => { const content = "if (x) {\n oldBody();\n}\nafter();"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_range", pos: makeTag(2, " oldBody();"), end: makeTag(3, "}"), lines: ["if (x) {", " newBody();", "}"], @@ -569,7 +573,9 @@ describe("applyHashlineEdits — heuristics", () => { delete Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS; try { const content = "root\n\tchild\n\t\tvalue\nend"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(3, "\t\tvalue"), lines: ["\\t\\treplaced"] }]; + const edits: HashlineEdit[] = [ + { op: "replace_line", pos: makeTag(3, "\t\tvalue"), lines: ["\\t\\treplaced"] }, + ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("root\n\tchild\n\t\treplaced\nend"); expect(result.warnings).toHaveLength(1); @@ -585,7 +591,9 @@ describe("applyHashlineEdits — heuristics", () => { Bun.env.PI_HASHLINE_AUTOCORRECT_ESCAPED_TABS = "0"; try { const content = "root\n\tchild\n\t\tvalue\nend"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(3, "\t\tvalue"), lines: ["\\t\\treplaced"] }]; + const edits: HashlineEdit[] = [ + { op: "replace_line", pos: makeTag(3, "\t\tvalue"), lines: ["\\t\\treplaced"] }, + ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("root\n\tchild\n\\t\\treplaced\nend"); expect(result.warnings).toBeUndefined(); @@ -602,7 +610,7 @@ describe("applyHashlineEdits — heuristics", () => { const content = "root\n\tchild\n\t\tvalue\nend"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_line", pos: makeTag(3, "\t\tvalue"), lines: ["\t\talready-tab", "\\t\\tescaped-still-literal"], }, @@ -618,7 +626,7 @@ describe("applyHashlineEdits — heuristics", () => { it("warns on literal \\uDDDD without changing content", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "bbb"), lines: ["\\uDDDD"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(2, "bbb"), lines: ["\\uDDDD"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\n\\uDDDD\nccc"); expect(result.warnings).toHaveLength(1); @@ -634,8 +642,8 @@ describe("applyHashlineEdits — multiple edits", () => { it("applies two non-overlapping replaces (bottom-up safe)", () => { const content = "aaa\nbbb\nccc\nddd\neee"; const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "bbb"), lines: ["BBB"] }, - { op: "replace", pos: makeTag(4, "ddd"), lines: ["DDD"] }, + { op: "replace_line", pos: makeTag(2, "bbb"), lines: ["BBB"] }, + { op: "replace_line", pos: makeTag(4, "ddd"), lines: ["DDD"] }, ]; const result = applyHashlineEdits(content, edits); @@ -646,8 +654,8 @@ describe("applyHashlineEdits — multiple edits", () => { it("applies replace + delete in one call", () => { const content = "aaa\nbbb\nccc\nddd"; const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(2, "bbb"), lines: ["BBB"] }, - { op: "replace", pos: makeTag(4, "ddd"), lines: [] }, + { op: "replace_line", pos: makeTag(2, "bbb"), lines: ["BBB"] }, + { op: "replace_line", pos: makeTag(4, "ddd"), lines: [] }, ]; const result = applyHashlineEdits(content, edits); @@ -657,7 +665,7 @@ describe("applyHashlineEdits — multiple edits", () => { it("applies replace + append in one call", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [ - { op: "replace", pos: makeTag(3, "ccc"), lines: ["CCC"] }, + { op: "replace_line", pos: makeTag(3, "ccc"), lines: ["CCC"] }, { op: "append", pos: makeTag(1, "aaa"), lines: ["INSERTED"] }, ]; @@ -669,12 +677,12 @@ describe("applyHashlineEdits — multiple edits", () => { const content = "one\ntwo\nthree\nfour\nfive\nsix"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_range", pos: makeTag(2, "two"), end: makeTag(3, "three"), lines: ["TWO_THREE"], }, - { op: "replace", pos: makeTag(6, "six"), lines: ["SIX"] }, + { op: "replace_line", pos: makeTag(6, "six"), lines: ["SIX"] }, ]; const result = applyHashlineEdits(content, edits); @@ -684,7 +692,9 @@ describe("applyHashlineEdits — multiple edits", () => { it("single-line replace expanding to multiple lines is not a noop", () => { const content = "aaa\n\nccc"; const blankHash = computeLineHash(2, ""); - const edits: HashlineEdit[] = [{ op: "replace", pos: { line: 2, hash: blankHash }, lines: ["", "inserted", ""] }]; + const edits: HashlineEdit[] = [ + { op: "replace_line", pos: { line: 2, hash: blankHash }, lines: ["", "inserted", ""] }, + ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\n\ninserted\n\nccc"); expect(result.firstChangedLine).toBe(2); @@ -706,13 +716,13 @@ describe("applyHashlineEdits — errors", () => { it("rejects stale hash", () => { const content = "aaa\nbbb\nccc"; // Use a hash that doesn't match any line (avoid 00 — ccc hashes to 00) - const edits: HashlineEdit[] = [{ op: "replace", pos: parseTag("2#QQ"), lines: ["BBB"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: parseTag("2#QQ"), lines: ["BBB"] }]; expect(() => applyHashlineEdits(content, edits)).toThrow(HashlineMismatchError); }); it("stale hash error shows >>> markers with correct hashes", () => { const content = "aaa\nbbb\nccc\nddd\neee"; - const edits: HashlineEdit[] = [{ op: "replace", pos: parseTag("2#QQ"), lines: ["BBB"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: parseTag("2#QQ"), lines: ["BBB"] }]; try { applyHashlineEdits(content, edits); @@ -736,8 +746,8 @@ describe("applyHashlineEdits — errors", () => { const content = "aaa\nbbb\nccc\nddd\neee"; // Use hashes that don't match any line (avoid 00 — ccc hashes to 00) const edits: HashlineEdit[] = [ - { op: "replace", pos: parseTag("2#ZZ"), lines: ["BBB"] }, - { op: "replace", pos: parseTag("4#ZZ"), lines: ["DDD"] }, + { op: "replace_line", pos: parseTag("2#ZZ"), lines: ["BBB"] }, + { op: "replace_line", pos: parseTag("4#ZZ"), lines: ["DDD"] }, ]; try { @@ -758,7 +768,7 @@ describe("applyHashlineEdits — errors", () => { it("does not relocate stale line refs even when hash uniquely matches another line", () => { const content = "aaa\nbbb\nccc"; const staleButUnique = parseTag(`2#${computeLineHash(1, "ccc")}`); - const edits: HashlineEdit[] = [{ op: "replace", pos: staleButUnique, lines: ["CCC"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: staleButUnique, lines: ["CCC"] }]; try { applyHashlineEdits(content, edits); expect.unreachable("should have thrown"); @@ -772,21 +782,23 @@ describe("applyHashlineEdits — errors", () => { it("does not relocate when expected hash is non-unique", () => { const content = "dup\nmid\ndup"; const staleDuplicate = parseTag(`2#${computeLineHash(1, "dup")}`); - const edits: HashlineEdit[] = [{ op: "replace", pos: staleDuplicate, lines: ["DUP"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: staleDuplicate, lines: ["DUP"] }]; expect(() => applyHashlineEdits(content, edits)).toThrow(HashlineMismatchError); }); it("rejects out-of-range line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "replace", pos: parseTag("10#ZZ"), lines: ["X"] }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: parseTag("10#ZZ"), lines: ["X"] }]; expect(() => applyHashlineEdits(content, edits)).toThrow(/does not exist/); }); it("rejects range with start > end", () => { const content = "aaa\nbbb\nccc\nddd\neee"; - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(5, "eee"), end: makeTag(2, "bbb"), lines: ["X"] }]; + const edits: HashlineEdit[] = [ + { op: "replace_range", pos: makeTag(5, "eee"), end: makeTag(2, "bbb"), lines: ["X"] }, + ]; expect(() => applyHashlineEdits(content, edits)).toThrow(); }); @@ -977,7 +989,7 @@ describe("hashlineParseContent", () => { const fileContent = "# Title\n- old item\n- old item 2\nfooter"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_line", pos: makeTag(2, "- old item"), lines: hashlineParseText("- [x] new item"), }, @@ -990,7 +1002,7 @@ describe("hashlineParseContent", () => { // All replacement lines start with '- ', triggering the 50% heuristic when '-' matched. const fileContent = "- [x] done\n- [ ] pending\n- [ ] also pending"; const newContent = hashlineParseText("- [x] done"); - const edits: HashlineEdit[] = [{ op: "replace", pos: makeTag(2, "- [ ] pending"), lines: newContent }]; + const edits: HashlineEdit[] = [{ op: "replace_line", pos: makeTag(2, "- [ ] pending"), lines: newContent }]; const result = applyHashlineEdits(fileContent, edits); expect(result.lines).toBe("- [x] done\n- [x] done\n- [ ] also pending"); }); @@ -1014,7 +1026,7 @@ describe("hashlineParseContent", () => { const fileContent = [" # cuDNN section", " # Note: Using version 1.23.0", ' $Version = "1.23.0"'].join("\n"); const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_line", pos: makeTag(2, " # Note: Using version 1.23.0"), lines: hashlineParseText([" # Note: Using version 1.24.x"]), }, @@ -1029,7 +1041,7 @@ describe("hashlineParseContent", () => { const fileContent = "const x = 1;\n// TODO: old\n# TODO: remove this\nconst y = 2;"; const edits: HashlineEdit[] = [ { - op: "replace", + op: "replace_line", pos: makeTag(3, "# TODO: remove this"), lines: hashlineParseText(["# TODO: remove this -- done"]), }, diff --git a/packages/coding-agent/test/initial-message.test.ts b/packages/coding-agent/test/initial-message.test.ts new file mode 100644 index 000000000..96fc9c6e3 --- /dev/null +++ b/packages/coding-agent/test/initial-message.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from "bun:test"; +import type { ImageContent } from "@oh-my-pi/pi-ai"; +import type { Args } from "../src/cli/args"; +import { buildInitialMessage } from "../src/cli/initial-message"; + +function createArgs(messages: string[]): Args { + return { + messages, + fileArgs: [], + unknownFlags: new Map(), + }; +} + +describe("buildInitialMessage", () => { + it("combines stdin, file text, and the first CLI message", () => { + const parsed = createArgs(["first", "second"]); + const images: ImageContent[] = [{ type: "image", data: "abc123", mimeType: "image/png" }]; + + const result = buildInitialMessage({ + parsed, + stdinContent: "stdin", + fileText: "file-", + fileImages: images, + }); + + expect(result.initialMessage).toBe("stdin\nfile-first"); + expect(result.initialImages).toEqual(images); + expect(parsed.messages).toEqual(["second"]); + }); + + it("leaves plain CLI messages untouched when there is no initial file or stdin input", () => { + const parsed = createArgs(["first", "second"]); + + const result = buildInitialMessage({ parsed }); + + expect(result.initialMessage).toBeUndefined(); + expect(result.initialImages).toBeUndefined(); + expect(parsed.messages).toEqual(["first", "second"]); + }); +}); diff --git a/packages/coding-agent/test/keybindings-display.test.ts b/packages/coding-agent/test/keybindings-display.test.ts index 8b1319271..1a0e37bd4 100644 --- a/packages/coding-agent/test/keybindings-display.test.ts +++ b/packages/coding-agent/test/keybindings-display.test.ts @@ -4,25 +4,25 @@ import { KeybindingsManager } from "../src/config/keybindings"; describe("KeybindingsManager.getDisplayString", () => { it("formats a single binding as a human-readable key hint", () => { const keybindings = KeybindingsManager.inMemory({ - dequeue: "alt+up", + "app.message.dequeue": "alt+up", }); - expect(keybindings.getDisplayString("dequeue")).toBe("Alt+Up"); + expect(keybindings.getDisplayString("app.message.dequeue")).toBe("Alt+Up"); }); it("formats multiple bindings with the existing separator", () => { const keybindings = KeybindingsManager.inMemory({ - copyPrompt: ["alt+shift+c", "ctrl+shift+c"], + "app.clipboard.copyPrompt": ["alt+shift+c", "ctrl+shift+c"], }); - expect(keybindings.getDisplayString("copyPrompt")).toBe("Alt+Shift+C/Ctrl+Shift+C"); + expect(keybindings.getDisplayString("app.clipboard.copyPrompt")).toBe("Alt+Shift+C/Ctrl+Shift+C"); }); it("returns an empty string when the action has no binding", () => { const keybindings = KeybindingsManager.inMemory({ - copyPrompt: [], + "app.clipboard.copyPrompt": [], }); - expect(keybindings.getDisplayString("copyPrompt")).toBe(""); + expect(keybindings.getDisplayString("app.clipboard.copyPrompt")).toBe(""); }); }); diff --git a/packages/coding-agent/test/keybindings-migration.test.ts b/packages/coding-agent/test/keybindings-migration.test.ts new file mode 100644 index 000000000..4a83ba06b --- /dev/null +++ b/packages/coding-agent/test/keybindings-migration.test.ts @@ -0,0 +1,50 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { setKeybindings } from "@oh-my-pi/pi-tui"; +import { KeybindingsManager } from "../src/config/keybindings"; + +describe("KeybindingsManager.create", () => { + beforeEach(() => { + setKeybindings(KeybindingsManager.inMemory()); + }); + + afterEach(() => { + setKeybindings(KeybindingsManager.inMemory()); + }); + + it("migrates legacy keybinding names on disk during create", async () => { + const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-")); + const configPath = path.join(agentDir, "keybindings.json"); + + await Bun.write( + configPath, + `${JSON.stringify( + { + fork: "ctrl+f", + selectConfirm: "enter", + cursorUp: "ctrl+p", + }, + null, + 2, + )}\n`, + ); + + try { + const manager = KeybindingsManager.create(agentDir); + const writtenConfig = await Bun.file(configPath).json(); + + expect(manager.getKeys("app.session.fork")).toEqual(["ctrl+f"]); + expect(manager.getKeys("tui.select.confirm")).toEqual(["enter"]); + expect(manager.getKeys("tui.editor.cursorUp")).toEqual(["ctrl+p"]); + expect(writtenConfig).toEqual({ + "app.session.fork": "ctrl+f", + "tui.editor.cursorUp": "ctrl+p", + "tui.select.confirm": "enter", + }); + } finally { + await fs.rm(agentDir, { recursive: true, force: true }); + } + }); +}); diff --git a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts index 0a74b0d22..cecd42a47 100644 --- a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts +++ b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts @@ -4,23 +4,23 @@ import { buildHotkeysMarkdown } from "../../../src/modes/utils/hotkeys-markdown" describe("buildHotkeysMarkdown", () => { it("emits flush-left markdown so headings and tables are parsed instead of treated as indented text", () => { const displayStrings: Record = { - copyLine: "Alt+Shift+L", - copyPrompt: "Ctrl+Shift+P", - togglePlanMode: "Alt+M", - expandTools: "Ctrl+O", - interrupt: "Esc", - clear: "Ctrl+C", - exit: "Ctrl+D", - suspend: "Ctrl+Z", - cycleThinkingLevel: "Shift+Tab", - cycleModelForward: "Ctrl+P", - cycleModelBackward: "Shift+Ctrl+P", - selectModel: "Ctrl+L", - historySearch: "Ctrl+R", - toggleThinking: "Ctrl+T", - externalEditor: "Ctrl+G", - pasteImage: "Ctrl+V", - toggleSTT: "Alt+H", + "app.clipboard.copyLine": "Alt+Shift+L", + "app.clipboard.copyPrompt": "Ctrl+Shift+P", + "app.plan.toggle": "Alt+M", + "app.tools.expand": "Ctrl+O", + "app.interrupt": "Esc", + "app.clear": "Ctrl+C", + "app.exit": "Ctrl+D", + "app.suspend": "Ctrl+Z", + "app.thinking.cycle": "Shift+Tab", + "app.model.cycleForward": "Ctrl+P", + "app.model.cycleBackward": "Shift+Ctrl+P", + "app.model.select": "Ctrl+L", + "app.history.search": "Ctrl+R", + "app.thinking.toggle": "Ctrl+T", + "app.editor.external": "Ctrl+G", + "app.clipboard.pasteImage": "Ctrl+V", + "app.stt.toggle": "Alt+H", }; const markdown = buildHotkeysMarkdown({ keybindings: { diff --git a/packages/coding-agent/test/prompt-action-autocomplete.test.ts b/packages/coding-agent/test/prompt-action-autocomplete.test.ts index 8dda96782..8a075b8d5 100644 --- a/packages/coding-agent/test/prompt-action-autocomplete.test.ts +++ b/packages/coding-agent/test/prompt-action-autocomplete.test.ts @@ -1,30 +1,30 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { EditorKeybindingsManager, setEditorKeybindings } from "../../tui/src/keybindings"; -import { KeybindingsManager } from "../src/config/keybindings"; +import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "@oh-my-pi/pi-tui"; +import { KeybindingsManager as AppKeybindingsManager } from "../src/config/keybindings"; import { createPromptActionAutocompleteProvider } from "../src/modes/prompt-action-autocomplete"; describe("prompt action autocomplete", () => { beforeEach(() => { - setEditorKeybindings( - new EditorKeybindingsManager({ - cursorLineStart: ["home", "f6"], - cursorLineEnd: "f7", - undo: "f8", + setKeybindings( + new KeybindingsManager({ + "tui.editor.cursorLineStart": { defaultKeys: ["home", "f6"], description: "Move cursor to line start" }, + "tui.editor.cursorLineEnd": { defaultKeys: "f7", description: "Move cursor to line end" }, + "tui.editor.undo": { defaultKeys: "f8", description: "Undo" }, }), ); }); afterEach(() => { - setEditorKeybindings(new EditorKeybindingsManager()); + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); }); it("shows prompt actions with configured shortcut hints", async () => { const provider = createPromptActionAutocompleteProvider({ commands: [], basePath: "/tmp", - keybindings: KeybindingsManager.inMemory({ - copyLine: "ctrl+shift+l", - copyPrompt: ["alt+shift+c", "ctrl+shift+c"], + keybindings: AppKeybindingsManager.inMemory({ + "app.clipboard.copyLine": "ctrl+shift+l", + "app.clipboard.copyPrompt": ["alt+shift+c", "ctrl+shift+c"], }), copyCurrentLine: () => {}, copyPrompt: () => {}, @@ -64,7 +64,7 @@ describe("prompt action autocomplete", () => { const provider = createPromptActionAutocompleteProvider({ commands: [], basePath: "/tmp", - keybindings: KeybindingsManager.inMemory(), + keybindings: AppKeybindingsManager.inMemory(), copyCurrentLine: () => {}, copyPrompt: () => {}, undo: prefix => { @@ -97,7 +97,7 @@ describe("prompt action autocomplete", () => { const provider = createPromptActionAutocompleteProvider({ commands: [], basePath: "/tmp", - keybindings: KeybindingsManager.inMemory(), + keybindings: AppKeybindingsManager.inMemory(), copyCurrentLine: () => {}, copyPrompt: () => {}, undo: () => {}, diff --git a/packages/tui/src/components/cancellable-loader.ts b/packages/tui/src/components/cancellable-loader.ts index 46317ed4c..c82bfd574 100644 --- a/packages/tui/src/components/cancellable-loader.ts +++ b/packages/tui/src/components/cancellable-loader.ts @@ -1,4 +1,4 @@ -import { matchesKey } from "../keys"; +import { getKeybindings } from "../keybindings"; import { Loader } from "./loader"; /** @@ -27,7 +27,8 @@ export class CancellableLoader extends Loader { } handleInput(data: string): void { - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + const kb = getKeybindings(); + if (kb.matches(data, "tui.select.cancel")) { this.#abortController.abort(); this.onAbort?.(); } diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 2840e1b7a..3095d6f3a 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -1,7 +1,7 @@ import { getProjectDir } from "@oh-my-pi/pi-utils"; import type { AutocompleteProvider, CombinedAutocompleteProvider } from "../autocomplete"; import { BracketedPasteHandler } from "../bracketed-paste"; -import { type EditorKeybindingsManager, getEditorKeybindings } from "../keybindings"; +import { getKeybindings, type KeybindingsManager } from "../keybindings"; import { extractPrintableText, matchesKey } from "../keys"; import { KillRing } from "../kill-ring"; import type { SymbolTheme } from "../symbols"; @@ -15,7 +15,12 @@ import { truncateToWidth, visibleWidth, } from "../utils"; -import { SelectList, type SelectListTheme } from "./select-list"; +import { SelectList, type SelectListLayoutOptions, type SelectListTheme } from "./select-list"; + +const SLASH_COMMAND_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 32, +}; const segmenter = getSegmenter(); @@ -691,12 +696,12 @@ export class Editor implements Component, Focusable { } handleInput(data: string): void { - const kb = getEditorKeybindings(); + const kb = getKeybindings(); // Handle character jump mode (awaiting next character to jump to) if (this.#jumpMode !== null) { // Cancel if the hotkey is pressed again - if (kb.matches(data, "jumpForward") || kb.matches(data, "jumpBackward")) { + if (kb.matches(data, "tui.editor.jumpForward") || kb.matches(data, "tui.editor.jumpBackward")) { this.#jumpMode = null; return; } @@ -728,12 +733,12 @@ export class Editor implements Component, Focusable { // Handle special key combinations first // Ctrl+C - Exit (let parent handle this) - if (matchesKey(data, "ctrl+c")) { + if (kb.matches(data, "tui.input.copy")) { return; } // Undo - if (kb.matches(data, "undo")) { + if (kb.matches(data, "tui.editor.undo")) { this.#applyUndo(); return; } @@ -741,27 +746,26 @@ export class Editor implements Component, Focusable { // Handle autocomplete special keys first (but don't block other input) if (this.#autocompleteState && this.#autocompleteList) { // Escape - cancel autocomplete - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (kb.matches(data, "tui.select.cancel")) { this.#cancelAutocomplete(true); return; } // Let the autocomplete list handle navigation and selection else if ( - matchesKey(data, "up") || - matchesKey(data, "down") || - matchesKey(data, "pageUp") || - matchesKey(data, "pageDown") || - matchesKey(data, "enter") || - matchesKey(data, "return") || + kb.matches(data, "tui.select.up") || + kb.matches(data, "tui.select.down") || + kb.matches(data, "tui.select.pageUp") || + kb.matches(data, "tui.select.pageDown") || + kb.matches(data, "tui.input.submit") || data === "\n" || - matchesKey(data, "tab") + kb.matches(data, "tui.input.tab") ) { // Only pass navigation keys to the list, not Enter/Tab (we handle those directly) if ( - matchesKey(data, "up") || - matchesKey(data, "down") || - matchesKey(data, "pageUp") || - matchesKey(data, "pageDown") + kb.matches(data, "tui.select.up") || + kb.matches(data, "tui.select.down") || + kb.matches(data, "tui.select.pageUp") || + kb.matches(data, "tui.select.pageDown") ) { this.#autocompleteList.handleInput(data); this.onAutocompleteUpdate?.(); @@ -769,7 +773,7 @@ export class Editor implements Component, Focusable { } // If Tab was pressed, always apply the selection - if (matchesKey(data, "tab")) { + if (kb.matches(data, "tui.input.tab")) { const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const shouldChainSlashCommandAutocomplete = this.#isSlashCommandNameAutocompleteSelection(); @@ -801,10 +805,7 @@ export class Editor implements Component, Focusable { } // If Enter was pressed on a slash command, apply completion and submit - if ( - (matchesKey(data, "enter") || matchesKey(data, "return") || data === "\n") && - this.#autocompletePrefix.startsWith("/") - ) { + if ((kb.matches(data, "tui.input.submit") || data === "\n") && this.#autocompletePrefix.startsWith("/")) { // Check for stale autocomplete state due to debounce const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); @@ -832,7 +833,7 @@ export class Editor implements Component, Focusable { // Don't return - fall through to submission logic } // If Enter was pressed on a file path, apply completion - else if (matchesKey(data, "enter") || matchesKey(data, "return") || data === "\n") { + else if (kb.matches(data, "tui.input.submit") || data === "\n") { const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const result = this.#autocompleteProvider.applyCompletion( @@ -863,7 +864,7 @@ export class Editor implements Component, Focusable { } // Tab key - context-aware completion (but not when already autocompleting) - if (matchesKey(data, "tab") && !this.#autocompleteState) { + if (kb.matches(data, "tui.input.tab") && !this.#autocompleteState) { this.#handleTabCompletion(); return; } @@ -920,7 +921,7 @@ export class Editor implements Component, Focusable { data === "\x1b[27;5;13~" || // Ctrl+Enter (legacy format) data === "\x1b\r" || // Option+Enter in some terminals (legacy) data === "\x1b[13;2~" || // Shift+Enter in some terminals (legacy format) - matchesKey(data, "shift+enter") || // Shift+Enter (Kitty protocol, handles lock bits) + kb.matches(data, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits) (data.length > 1 && data.includes("\x1b") && data.includes("\r")) || (data === "\n" && data.length === 1) // Shift+Enter from iTerm2 mapping ) { @@ -932,7 +933,7 @@ export class Editor implements Component, Focusable { this.#addNewLine(); } // Plain Enter - submit (handles both legacy \r and Kitty protocol with lock bits) - else if (matchesKey(data, "enter") || matchesKey(data, "return") || data === "\n") { + else if (kb.matches(data, "tui.input.submit") || data === "\n") { // If submit is disabled, do nothing if (this.disableSubmit) { return; @@ -941,17 +942,17 @@ export class Editor implements Component, Focusable { this.#submitValue(); } // Backspace (including Shift+Backspace) - else if (matchesKey(data, "backspace") || matchesKey(data, "shift+backspace")) { + else if (kb.matches(data, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) { this.#handleBackspace(); } // Line navigation shortcuts (Home/End keys) - else if (matchesKey(data, "home")) { + else if (kb.matches(data, "tui.editor.cursorLineStart")) { this.#moveToLineStart(); - } else if (matchesKey(data, "end")) { + } else if (kb.matches(data, "tui.editor.cursorLineEnd")) { this.#moveToLineEnd(); } // Page navigation (PageUp/PageDown) - else if (matchesKey(data, "pageUp")) { + else if (kb.matches(data, "tui.editor.pageUp")) { if (this.#isEditorEmpty()) { this.#navigateHistory(-1); } else if (this.#historyIndex > -1 && this.#isOnFirstVisualLine()) { @@ -959,7 +960,7 @@ export class Editor implements Component, Focusable { } else { this.#pageScroll(-1); } - } else if (matchesKey(data, "pageDown")) { + } else if (kb.matches(data, "tui.editor.pageDown")) { if (this.#historyIndex > -1 && this.#isOnLastVisualLine()) { this.#navigateHistory(1); } else { @@ -967,21 +968,21 @@ export class Editor implements Component, Focusable { } } // Forward delete (Fn+Backspace or Delete key, including Shift+Delete) - else if (matchesKey(data, "delete") || matchesKey(data, "shift+delete")) { + else if (kb.matches(data, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) { this.#handleForwardDelete(); } // Word navigation (Option/Alt + Arrow or Ctrl + Arrow) - else if (matchesKey(data, "alt+left") || matchesKey(data, "ctrl+left")) { + else if (kb.matches(data, "tui.editor.cursorWordLeft")) { // Word left this.#resetKillSequence(); this.#moveWordBackwards(); - } else if (matchesKey(data, "alt+right") || matchesKey(data, "ctrl+right")) { + } else if (kb.matches(data, "tui.editor.cursorWordRight")) { // Word right this.#resetKillSequence(); this.#moveWordForwards(); } // Arrow keys - else if (matchesKey(data, "up")) { + else if (kb.matches(data, "tui.editor.cursorUp")) { // Up - history navigation or cursor movement if (this.#isEditorEmpty()) { this.#navigateHistory(-1); // Start browsing history @@ -993,7 +994,7 @@ export class Editor implements Component, Focusable { } else { this.#moveCursor(-1, 0); // Cursor movement (within text or history entry) } - } else if (matchesKey(data, "down")) { + } else if (kb.matches(data, "tui.editor.cursorDown")) { // Down - history navigation or cursor movement if (this.#historyIndex > -1 && this.#isOnLastVisualLine()) { this.#navigateHistory(1); // Navigate to newer history entry or clear @@ -1003,10 +1004,10 @@ export class Editor implements Component, Focusable { } else { this.#moveCursor(1, 0); // Cursor movement (within text or history entry) } - } else if (matchesKey(data, "right")) { + } else if (kb.matches(data, "tui.editor.cursorRight")) { // Right this.#moveCursor(0, 1); - } else if (matchesKey(data, "left")) { + } else if (kb.matches(data, "tui.editor.cursorLeft")) { // Left this.#moveCursor(0, -1); } @@ -1015,9 +1016,9 @@ export class Editor implements Component, Focusable { this.#insertCharacter(" "); } // Character jump mode triggers - else if (kb.matches(data, "jumpForward")) { + else if (kb.matches(data, "tui.editor.jumpForward")) { this.#jumpMode = "forward"; - } else if (kb.matches(data, "jumpBackward")) { + } else if (kb.matches(data, "tui.editor.jumpBackward")) { this.#jumpMode = "backward"; } // Printable keystrokes, including Kitty CSI-u text-producing sequences. @@ -1393,10 +1394,10 @@ export class Editor implements Component, Focusable { } } - #shouldSubmitOnBackslashEnter(data: string, kb: EditorKeybindingsManager): boolean { + #shouldSubmitOnBackslashEnter(data: string, kb: KeybindingsManager): boolean { if (this.disableSubmit) return false; if (!matchesKey(data, "enter")) return false; - const submitKeys = kb.getKeys("submit"); + const submitKeys = kb.getKeys("tui.input.submit"); const hasShiftEnter = submitKeys.includes("shift+enter") || submitKeys.includes("shift+return"); if (!hasShiftEnter) return false; @@ -2184,11 +2185,7 @@ export class Editor implements Component, Focusable { if (suggestions && suggestions.items.length > 0) { this.#autocompletePrefix = suggestions.prefix; - this.#autocompleteList = new SelectList( - suggestions.items, - this.#autocompleteMaxVisible, - this.#theme.selectList, - ); + this.#autocompleteList = this.#createAutocompleteList(suggestions.prefix, suggestions.items); this.#autocompleteState = "regular"; this.onAutocompleteUpdate?.(); } else { @@ -2196,6 +2193,16 @@ export class Editor implements Component, Focusable { this.onAutocompleteUpdate?.(); } } + #createAutocompleteList( + prefix: string, + items: Array<{ value: string; label: string; description?: string }>, + ): SelectList { + // Layout options prepared for future SelectList enhancements (e.g., for slash commands) + const layout = prefix.startsWith("/") ? SLASH_COMMAND_SELECT_LIST_LAYOUT : undefined; + // TODO: Pass layout to SelectList when constructor is updated to support it + void layout; // Use layout variable to avoid lint warnings + return new SelectList(items, this.#autocompleteMaxVisible, this.#theme.selectList); + } #handleTabCompletion(): void { if (!this.#autocompleteProvider) return; @@ -2263,11 +2270,7 @@ https://github.com/EsotericSoftware/spine-runtimes/actions/runs/19536643416/job/ } this.#autocompletePrefix = suggestions.prefix; - this.#autocompleteList = new SelectList( - suggestions.items, - this.#autocompleteMaxVisible, - this.#theme.selectList, - ); + this.#autocompleteList = this.#createAutocompleteList(suggestions.prefix, suggestions.items); this.#autocompleteState = "force"; this.onAutocompleteUpdate?.(); } else { @@ -2313,11 +2316,7 @@ https://github.com/EsotericSoftware/spine-runtimes/actions/runs/19536643416/job/ if (suggestions && suggestions.items.length > 0) { this.#autocompletePrefix = suggestions.prefix; // Always create new SelectList to ensure update - this.#autocompleteList = new SelectList( - suggestions.items, - this.#autocompleteMaxVisible, - this.#theme.selectList, - ); + this.#autocompleteList = this.#createAutocompleteList(suggestions.prefix, suggestions.items); this.onAutocompleteUpdate?.(); } else { this.#cancelAutocomplete(); diff --git a/packages/tui/src/components/input.ts b/packages/tui/src/components/input.ts index fc7278eb8..e9ff9b721 100644 --- a/packages/tui/src/components/input.ts +++ b/packages/tui/src/components/input.ts @@ -1,5 +1,5 @@ import { BracketedPasteHandler } from "../bracketed-paste"; -import { getEditorKeybindings } from "../keybindings"; +import { getKeybindings } from "../keybindings"; import { extractPrintableText } from "../keys"; import { KillRing } from "../kill-ring"; import { type Component, CURSOR_MARKER, type Focusable } from "../tui"; @@ -65,69 +65,69 @@ export class Input implements Component, Focusable { return; } - const kb = getEditorKeybindings(); + const kb = getKeybindings(); // Escape/Cancel - if (kb.matches(data, "selectCancel")) { + if (kb.matches(data, "tui.select.cancel")) { if (this.onEscape) this.onEscape(); return; } // Undo - if (kb.matches(data, "undo")) { + if (kb.matches(data, "tui.editor.undo")) { this.#undo(); return; } // Submit - if (kb.matches(data, "submit") || data === "\n") { + if (kb.matches(data, "tui.input.submit") || data === "\n") { if (this.onSubmit) this.onSubmit(this.#value); return; } // Deletion - if (kb.matches(data, "deleteCharBackward")) { + if (kb.matches(data, "tui.editor.deleteCharBackward")) { this.#handleBackspace(); return; } - if (kb.matches(data, "deleteCharForward")) { + if (kb.matches(data, "tui.editor.deleteCharForward")) { this.#handleForwardDelete(); return; } - if (kb.matches(data, "deleteWordBackward")) { + if (kb.matches(data, "tui.editor.deleteWordBackward")) { this.#deleteWordBackwards(); return; } - if (kb.matches(data, "deleteWordForward")) { + if (kb.matches(data, "tui.editor.deleteWordForward")) { this.#deleteWordForward(); return; } - if (kb.matches(data, "deleteToLineStart")) { + if (kb.matches(data, "tui.editor.deleteToLineStart")) { this.#deleteToLineStart(); return; } - if (kb.matches(data, "deleteToLineEnd")) { + if (kb.matches(data, "tui.editor.deleteToLineEnd")) { this.#deleteToLineEnd(); return; } // Kill ring actions - if (kb.matches(data, "yank")) { + if (kb.matches(data, "tui.editor.yank")) { this.#yank(); return; } - if (kb.matches(data, "yankPop")) { + if (kb.matches(data, "tui.editor.yankPop")) { this.#yankPop(); return; } // Cursor movement - if (kb.matches(data, "cursorLeft")) { + if (kb.matches(data, "tui.editor.cursorLeft")) { this.#lastAction = null; if (this.#cursor > 0) { const beforeCursor = this.#value.slice(0, this.#cursor); @@ -138,7 +138,7 @@ export class Input implements Component, Focusable { return; } - if (kb.matches(data, "cursorRight")) { + if (kb.matches(data, "tui.editor.cursorRight")) { this.#lastAction = null; if (this.#cursor < this.#value.length) { const afterCursor = this.#value.slice(this.#cursor); @@ -149,24 +149,24 @@ export class Input implements Component, Focusable { return; } - if (kb.matches(data, "cursorLineStart")) { + if (kb.matches(data, "tui.editor.cursorLineStart")) { this.#lastAction = null; this.#cursor = 0; return; } - if (kb.matches(data, "cursorLineEnd")) { + if (kb.matches(data, "tui.editor.cursorLineEnd")) { this.#lastAction = null; this.#cursor = this.#value.length; return; } - if (kb.matches(data, "cursorWordLeft")) { + if (kb.matches(data, "tui.editor.cursorWordLeft")) { this.#moveWordBackwards(); return; } - if (kb.matches(data, "cursorWordRight")) { + if (kb.matches(data, "tui.editor.cursorWordRight")) { this.#moveWordForwards(); return; } diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index 2a8a32ee3..00c30ee41 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -306,7 +306,7 @@ export class Markdown implements Component { styledHeading = this.#theme.heading(this.#theme.bold(headingPrefix + headingText)); } lines.push(styledHeading); - if (nextTokenType !== "space") { + if (nextTokenType && nextTokenType !== "space") { lines.push(""); // Add spacing after headings (unless space token follows) } break; @@ -332,7 +332,7 @@ export class Markdown implements Component { for (const asciiLine of Bun.stripANSI(ascii).split("\n")) { lines.push(asciiLine); } - if (nextTokenType !== "space") { + if (nextTokenType && nextTokenType !== "space") { lines.push(""); } break; @@ -354,7 +354,7 @@ export class Markdown implements Component { } } lines.push(this.#theme.codeBlockBorder("```")); - if (nextTokenType !== "space") { + if (nextTokenType && nextTokenType !== "space") { lines.push(""); // Add spacing after code blocks (unless space token follows) } break; @@ -369,7 +369,7 @@ export class Markdown implements Component { } case "table": { - const tableLines = this.#renderTable(token as TableToken, width, styleContext); + const tableLines = this.#renderTable(token as TableToken, width, nextTokenType, styleContext); lines.push(...tableLines); break; } @@ -415,7 +415,7 @@ export class Markdown implements Component { lines.push(this.#theme.quoteBorder(`${this.#theme.symbols.quoteBorder} `) + wrappedLine); } } - if (nextTokenType !== "space") { + if (nextTokenType && nextTokenType !== "space") { lines.push(""); // Add spacing after blockquotes (unless space token follows) } break; @@ -423,7 +423,7 @@ export class Markdown implements Component { case "hr": lines.push(this.#theme.hr(this.#theme.symbols.hrChar.repeat(Math.min(width, 80)))); - if (nextTokenType !== "space") { + if (nextTokenType && nextTokenType !== "space") { lines.push(""); // Add spacing after horizontal rules (unless space token follows) } break; @@ -669,7 +669,12 @@ export class Markdown implements Component { * Render a table with width-aware cell wrapping. * Cells that don't fit are wrapped to multiple lines. */ - #renderTable(token: TableToken, availableWidth: number, styleContext?: InlineStyleContext): string[] { + #renderTable( + token: TableToken, + availableWidth: number, + nextTokenType?: string, + styleContext?: InlineStyleContext, + ): string[] { const lines: string[] = []; const numCols = token.header.length; @@ -684,7 +689,9 @@ export class Markdown implements Component { if (availableForCells < numCols) { // Too narrow to render a stable table. Fall back to raw markdown. const fallbackLines = token.raw ? wrapTextWithAnsi(token.raw, availableWidth) : []; - fallbackLines.push(""); + if (nextTokenType && nextTokenType !== "space") { + fallbackLines.push(""); + } return fallbackLines; } @@ -834,7 +841,9 @@ export class Markdown implements Component { const bottomBorderCells = columnWidths.map(w => h.repeat(w)); lines.push(`${t.bottomLeft}${h}${bottomBorderCells.join(`${h}${t.teeUp}${h}`)}${h}${t.bottomRight}`); - lines.push(""); // Add spacing after table + if (nextTokenType && nextTokenType !== "space") { + lines.push(""); // Add spacing after table + } return lines; } } diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index bfd5a8f0b..25ee62f0c 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -1,8 +1,21 @@ -import { matchesKey } from "../keys"; +import { getKeybindings } from "../keybindings"; import type { SymbolTheme } from "../symbols"; import type { Component } from "../tui"; import { Ellipsis, padding, replaceTabs, truncateToWidth, visibleWidth } from "../utils"; +const DEFAULT_PRIMARY_COLUMN_WIDTH = 32; +const PRIMARY_COLUMN_GAP = 2; +const MIN_DESCRIPTION_WIDTH = 10; + +function sanitizeSingleLine(text: string): string { + return replaceTabs(text) + .replace(/[\r\n]+/g, " ") + .replace(/\s+/g, " ") + .trim(); +} + +const clamp = (value: number, min: number, max: number): number => Math.max(min, Math.min(value, max)); + export interface SelectItem { value: string; label: string; @@ -20,11 +33,18 @@ export interface SelectListTheme { symbols: SymbolTheme; } -function sanitizeSingleLine(text: string): string { - return replaceTabs(text) - .replace(/[\r\n]+/g, " ") - .replace(/\s+/g, " ") - .trim(); +export interface SelectListTruncatePrimaryContext { + text: string; + maxWidth: number; + columnWidth: number; + item: SelectItem; + isSelected: boolean; +} + +export interface SelectListLayoutOptions { + minPrimaryColumnWidth?: number; + maxPrimaryColumnWidth?: number; + truncatePrimary?: (context: SelectListTruncatePrimaryContext) => string; } export class SelectList implements Component { @@ -39,6 +59,7 @@ export class SelectList implements Component { private readonly items: ReadonlyArray, private readonly maxVisible: number, private readonly theme: SelectListTheme, + private readonly layout: SelectListLayoutOptions = {}, ) { this.#filteredItems = items; } @@ -66,6 +87,8 @@ export class SelectList implements Component { return lines; } + const primaryColumnWidth = this.#getPrimaryColumnWidth(); + // Calculate visible range with scrolling const startIndex = Math.max( 0, @@ -79,71 +102,8 @@ export class SelectList implements Component { if (!item) continue; const isSelected = i === this.#selectedIndex; - const labelText = sanitizeSingleLine(item.label || item.value); const descriptionText = item.description ? sanitizeSingleLine(item.description) : undefined; - - let line = ""; - if (isSelected) { - // Use arrow indicator for selection - entire line uses selectedText color - const prefix = `${this.theme.symbols.cursor} `; - const prefixWidth = visibleWidth(prefix); - const displayValue = labelText; - - if (descriptionText && width > 40) { - // Calculate how much space we have for value + description - const maxValueWidth = Math.min(30, width - prefixWidth - 4); - const truncatedValue = truncateToWidth(displayValue, maxValueWidth, Ellipsis.Omit); - const spacing = padding(Math.max(1, 32 - truncatedValue.length)); - - // Calculate remaining space for description using visible widths - const descriptionStart = prefixWidth + truncatedValue.length + spacing.length; - const remainingWidth = width - descriptionStart - 2; // -2 for safety - - if (remainingWidth > 10) { - const truncatedDesc = truncateToWidth(descriptionText, remainingWidth, Ellipsis.Omit); - // Apply selectedText to entire line content - line = this.theme.selectedText(`${prefix}${truncatedValue}${spacing}${truncatedDesc}`); - } else { - // Not enough space for description - const maxWidth = width - prefixWidth - 2; - line = this.theme.selectedText(`${prefix}${truncateToWidth(displayValue, maxWidth, Ellipsis.Omit)}`); - } - } else { - // No description or not enough width - const maxWidth = width - prefixWidth - 2; - line = this.theme.selectedText(`${prefix}${truncateToWidth(displayValue, maxWidth, Ellipsis.Omit)}`); - } - } else { - const displayValue = labelText; - const prefix = padding(visibleWidth(this.theme.symbols.cursor) + 1); - - if (descriptionText && width > 40) { - // Calculate how much space we have for value + description - const maxValueWidth = Math.min(30, width - prefix.length - 4); - const truncatedValue = truncateToWidth(displayValue, maxValueWidth, Ellipsis.Omit); - const spacing = padding(Math.max(1, 32 - truncatedValue.length)); - - // Calculate remaining space for description - const descriptionStart = prefix.length + truncatedValue.length + spacing.length; - const remainingWidth = width - descriptionStart - 2; // -2 for safety - - if (remainingWidth > 10) { - const truncatedDesc = truncateToWidth(descriptionText, remainingWidth, Ellipsis.Omit); - const descText = this.theme.description(spacing + truncatedDesc); - line = prefix + truncatedValue + descText; - } else { - // Not enough space for description - const maxWidth = width - prefix.length - 2; - line = prefix + truncateToWidth(displayValue, maxWidth, Ellipsis.Omit); - } - } else { - // No description or not enough width - const maxWidth = width - prefix.length - 2; - line = prefix + truncateToWidth(displayValue, maxWidth, Ellipsis.Omit); - } - } - - lines.push(line); + lines.push(this.#renderItem(item, isSelected, width, descriptionText, primaryColumnWidth)); } // Add scroll indicators if needed @@ -158,41 +118,123 @@ export class SelectList implements Component { handleInput(keyData: string): void { if (this.#filteredItems.length === 0) return; + const kb = getKeybindings(); // Up arrow - wrap to bottom when at top - if (matchesKey(keyData, "up")) { + if (kb.matches(keyData, "tui.select.up")) { this.#selectedIndex = this.#selectedIndex === 0 ? this.#filteredItems.length - 1 : this.#selectedIndex - 1; this.#notifySelectionChange(); } // Down arrow - wrap to top when at bottom - else if (matchesKey(keyData, "down")) { + else if (kb.matches(keyData, "tui.select.down")) { this.#selectedIndex = this.#selectedIndex === this.#filteredItems.length - 1 ? 0 : this.#selectedIndex + 1; this.#notifySelectionChange(); } // PageUp - jump up by one visible page - else if (matchesKey(keyData, "pageUp")) { + else if (kb.matches(keyData, "tui.select.pageUp")) { this.#selectedIndex = Math.max(0, this.#selectedIndex - this.maxVisible); this.#notifySelectionChange(); } // PageDown - jump down by one visible page - else if (matchesKey(keyData, "pageDown")) { + else if (kb.matches(keyData, "tui.select.pageDown")) { this.#selectedIndex = Math.min(this.#filteredItems.length - 1, this.#selectedIndex + this.maxVisible); this.#notifySelectionChange(); } // Enter - else if (matchesKey(keyData, "enter") || matchesKey(keyData, "return") || keyData === "\n") { + else if (kb.matches(keyData, "tui.select.confirm") || keyData === "\n") { const selectedItem = this.#filteredItems[this.#selectedIndex]; if (selectedItem && this.onSelect) { this.onSelect(selectedItem); } } // Escape or Ctrl+C - else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc") || matchesKey(keyData, "ctrl+c")) { + else if (kb.matches(keyData, "tui.select.cancel")) { if (this.onCancel) { this.onCancel(); } } } + #renderItem( + item: SelectItem, + isSelected: boolean, + width: number, + descriptionSingleLine: string | undefined, + primaryColumnWidth: number, + ): string { + const prefix = isSelected + ? `${this.theme.symbols.cursor} ` + : padding(visibleWidth(this.theme.symbols.cursor) + 1); + const prefixWidth = visibleWidth(prefix); + + if (descriptionSingleLine && width > 40) { + const effectivePrimaryColumnWidth = Math.max(1, Math.min(primaryColumnWidth, width - prefixWidth - 4)); + const maxPrimaryWidth = Math.max(1, effectivePrimaryColumnWidth - PRIMARY_COLUMN_GAP); + const truncatedValue = this.#truncatePrimary(item, isSelected, maxPrimaryWidth, effectivePrimaryColumnWidth); + const truncatedValueWidth = visibleWidth(truncatedValue); + const spacing = padding(Math.max(1, effectivePrimaryColumnWidth - truncatedValueWidth)); + const descriptionStart = prefixWidth + truncatedValueWidth + spacing.length; + const remainingWidth = width - descriptionStart - 2; // -2 for safety + + if (remainingWidth > MIN_DESCRIPTION_WIDTH) { + const truncatedDesc = truncateToWidth(descriptionSingleLine, remainingWidth, Ellipsis.Omit); + if (isSelected) { + return this.theme.selectedText(`${prefix}${truncatedValue}${spacing}${truncatedDesc}`); + } + + const descText = this.theme.description(spacing + truncatedDesc); + return prefix + truncatedValue + descText; + } + } + + const maxWidth = width - prefixWidth - 2; + const truncatedValue = this.#truncatePrimary(item, isSelected, maxWidth, maxWidth); + if (isSelected) { + return this.theme.selectedText(`${prefix}${truncatedValue}`); + } + + return prefix + truncatedValue; + } + + #getPrimaryColumnWidth(): number { + const { min, max } = this.#getPrimaryColumnBounds(); + const widestPrimary = this.#filteredItems.reduce((widest, item) => { + return Math.max(widest, visibleWidth(this.#getDisplayValue(item)) + PRIMARY_COLUMN_GAP); + }, 0); + + return clamp(widestPrimary, min, max); + } + + #getPrimaryColumnBounds(): { min: number; max: number } { + const rawMin = + this.layout.minPrimaryColumnWidth ?? this.layout.maxPrimaryColumnWidth ?? DEFAULT_PRIMARY_COLUMN_WIDTH; + const rawMax = + this.layout.maxPrimaryColumnWidth ?? this.layout.minPrimaryColumnWidth ?? DEFAULT_PRIMARY_COLUMN_WIDTH; + + return { + min: Math.max(1, Math.min(rawMin, rawMax)), + max: Math.max(1, Math.max(rawMin, rawMax)), + }; + } + + #truncatePrimary(item: SelectItem, isSelected: boolean, maxWidth: number, columnWidth: number): string { + const displayValue = this.#getDisplayValue(item); + const truncatedValue = this.layout.truncatePrimary + ? this.layout.truncatePrimary({ + text: displayValue, + maxWidth, + columnWidth, + item, + isSelected, + }) + : truncateToWidth(displayValue, maxWidth, Ellipsis.Omit); + + return truncateToWidth(truncatedValue, maxWidth, Ellipsis.Omit); + } + + #getDisplayValue(item: SelectItem): string { + return sanitizeSingleLine(item.label || item.value); + } + #notifySelectionChange(): void { const selectedItem = this.#filteredItems[this.#selectedIndex]; if (selectedItem && this.onSelectionChange) { diff --git a/packages/tui/src/components/settings-list.ts b/packages/tui/src/components/settings-list.ts index dcf333d80..f705095e4 100644 --- a/packages/tui/src/components/settings-list.ts +++ b/packages/tui/src/components/settings-list.ts @@ -1,4 +1,4 @@ -import { matchesKey } from "../keys"; +import { getKeybindings } from "../keybindings"; import type { Component } from "../tui"; import { Ellipsis, padding, truncateToWidth, visibleWidth, wrapTextWithAnsi } from "../utils"; @@ -148,13 +148,14 @@ export class SettingsList implements Component { } // Main list input handling - if (matchesKey(data, "up")) { + const kb = getKeybindings(); + if (kb.matches(data, "tui.select.up")) { this.#selectedIndex = this.#selectedIndex === 0 ? this.#items.length - 1 : this.#selectedIndex - 1; - } else if (matchesKey(data, "down")) { + } else if (kb.matches(data, "tui.select.down")) { this.#selectedIndex = this.#selectedIndex === this.#items.length - 1 ? 0 : this.#selectedIndex + 1; - } else if (matchesKey(data, "enter") || matchesKey(data, "return") || data === "\n" || data === " ") { + } else if (kb.matches(data, "tui.select.confirm") || data === " " || data === "\n") { this.#activateItem(); - } else if (matchesKey(data, "escape") || matchesKey(data, "esc") || matchesKey(data, "ctrl+c")) { + } else if (kb.matches(data, "tui.select.cancel")) { this.#onCancel(); } } diff --git a/packages/tui/src/keybindings.ts b/packages/tui/src/keybindings.ts index 9daf484e4..47241210a 100644 --- a/packages/tui/src/keybindings.ts +++ b/packages/tui/src/keybindings.ts @@ -1,95 +1,145 @@ import { type KeyId, matchesKey, parseKey } from "./keys"; /** - * Editor actions that can be bound to keys. + * Global keybinding registry. + * Downstream packages can add keybindings via declaration merging. */ -export type EditorAction = - // Cursor movement - | "cursorUp" - | "cursorDown" - | "cursorLeft" - | "cursorRight" - | "cursorWordLeft" - | "cursorWordRight" - | "cursorLineStart" - | "cursorLineEnd" - | "jumpForward" - | "jumpBackward" - // Deletion - | "deleteCharBackward" - | "deleteCharForward" - | "deleteWordBackward" - | "deleteWordForward" - | "deleteToLineStart" - | "deleteToLineEnd" - // Text input - | "newLine" - | "submit" - | "tab" - // Selection/autocomplete - | "selectUp" - | "selectDown" - | "selectPageUp" - | "selectPageDown" - | "selectConfirm" - | "selectCancel" - // Clipboard - | "copy" - // Kill ring / undo - | "undo" - | "yank" - | "yankPop"; +export interface Keybindings { + // Editor navigation and editing + "tui.editor.cursorUp": true; + "tui.editor.cursorDown": true; + "tui.editor.cursorLeft": true; + "tui.editor.cursorRight": true; + "tui.editor.cursorWordLeft": true; + "tui.editor.cursorWordRight": true; + "tui.editor.cursorLineStart": true; + "tui.editor.cursorLineEnd": true; + "tui.editor.jumpForward": true; + "tui.editor.jumpBackward": true; + "tui.editor.pageUp": true; + "tui.editor.pageDown": true; + "tui.editor.deleteCharBackward": true; + "tui.editor.deleteCharForward": true; + "tui.editor.deleteWordBackward": true; + "tui.editor.deleteWordForward": true; + "tui.editor.deleteToLineStart": true; + "tui.editor.deleteToLineEnd": true; + "tui.editor.yank": true; + "tui.editor.yankPop": true; + "tui.editor.undo": true; + // Generic input actions + "tui.input.newLine": true; + "tui.input.submit": true; + "tui.input.tab": true; + "tui.input.copy": true; + // Generic selection actions + "tui.select.up": true; + "tui.select.down": true; + "tui.select.pageUp": true; + "tui.select.pageDown": true; + "tui.select.confirm": true; + "tui.select.cancel": true; +} + +export type Keybinding = keyof Keybindings; // Re-export KeyId from keys.ts export type { KeyId }; -/** - * Editor keybindings configuration. - */ -export type EditorKeybindingsConfig = { - [K in EditorAction]?: KeyId | KeyId[]; -}; +export interface KeybindingDefinition { + defaultKeys: KeyId | KeyId[]; + description?: string; +} -/** - * Default editor keybindings. - */ -export const DEFAULT_EDITOR_KEYBINDINGS: Required = { - // Cursor movement - cursorUp: "up", - cursorDown: "down", - cursorLeft: ["left", "ctrl+b"], - cursorRight: ["right", "ctrl+f"], - cursorWordLeft: ["alt+left", "ctrl+left", "alt+b"], - cursorWordRight: ["alt+right", "ctrl+right", "alt+f"], - cursorLineStart: ["home", "ctrl+a"], - cursorLineEnd: ["end", "ctrl+e"], - jumpForward: "ctrl+]", - jumpBackward: "ctrl+alt+]", - // Deletion - deleteCharBackward: "backspace", - deleteCharForward: ["delete", "ctrl+d"], - deleteWordBackward: ["ctrl+w", "alt+backspace", "ctrl+backspace"], - deleteWordForward: ["alt+delete", "alt+d"], - deleteToLineStart: "ctrl+u", - deleteToLineEnd: "ctrl+k", - // Text input - newLine: "shift+enter", - submit: "enter", - tab: "tab", - // Selection/autocomplete - selectUp: "up", - selectDown: "down", - selectPageUp: "pageUp", - selectPageDown: "pageDown", - selectConfirm: "enter", - selectCancel: ["escape", "ctrl+c"], - // Clipboard - copy: "ctrl+c", - // Kill ring / undo - undo: ["ctrl+-", "ctrl+_"], - yank: "ctrl+y", - yankPop: "alt+y", -}; +export type KeybindingDefinitions = Record; +export type KeybindingsConfig = Record; + +export const TUI_KEYBINDINGS = { + "tui.editor.cursorUp": { defaultKeys: "up", description: "Move cursor up" }, + "tui.editor.cursorDown": { defaultKeys: "down", description: "Move cursor down" }, + "tui.editor.cursorLeft": { + defaultKeys: ["left", "ctrl+b"], + description: "Move cursor left", + }, + "tui.editor.cursorRight": { + defaultKeys: ["right", "ctrl+f"], + description: "Move cursor right", + }, + "tui.editor.cursorWordLeft": { + defaultKeys: ["alt+left", "ctrl+left", "alt+b"], + description: "Move cursor word left", + }, + "tui.editor.cursorWordRight": { + defaultKeys: ["alt+right", "ctrl+right", "alt+f"], + description: "Move cursor word right", + }, + "tui.editor.cursorLineStart": { + defaultKeys: ["home", "ctrl+a"], + description: "Move to line start", + }, + "tui.editor.cursorLineEnd": { + defaultKeys: ["end", "ctrl+e"], + description: "Move to line end", + }, + "tui.editor.jumpForward": { + defaultKeys: "ctrl+]", + description: "Jump forward to character", + }, + "tui.editor.jumpBackward": { + defaultKeys: "ctrl+alt+]", + description: "Jump backward to character", + }, + "tui.editor.pageUp": { defaultKeys: "pageUp", description: "Page up" }, + "tui.editor.pageDown": { defaultKeys: "pageDown", description: "Page down" }, + "tui.editor.deleteCharBackward": { + defaultKeys: "backspace", + description: "Delete character backward", + }, + "tui.editor.deleteCharForward": { + defaultKeys: ["delete", "ctrl+d"], + description: "Delete character forward", + }, + "tui.editor.deleteWordBackward": { + defaultKeys: ["ctrl+w", "alt+backspace", "ctrl+backspace"], + description: "Delete word backward", + }, + "tui.editor.deleteWordForward": { + defaultKeys: ["alt+delete", "alt+d"], + description: "Delete word forward", + }, + "tui.editor.deleteToLineStart": { + defaultKeys: "ctrl+u", + description: "Delete to line start", + }, + "tui.editor.deleteToLineEnd": { + defaultKeys: "ctrl+k", + description: "Delete to line end", + }, + "tui.editor.yank": { defaultKeys: "ctrl+y", description: "Yank" }, + "tui.editor.yankPop": { defaultKeys: "alt+y", description: "Yank pop" }, + "tui.editor.undo": { defaultKeys: ["ctrl+-", "ctrl+_"], description: "Undo" }, + "tui.input.newLine": { defaultKeys: "shift+enter", description: "Insert newline" }, + "tui.input.submit": { defaultKeys: "enter", description: "Submit input" }, + "tui.input.tab": { defaultKeys: "tab", description: "Tab / autocomplete" }, + "tui.input.copy": { defaultKeys: "ctrl+c", description: "Copy selection" }, + "tui.select.up": { defaultKeys: "up", description: "Move selection up" }, + "tui.select.down": { defaultKeys: "down", description: "Move selection down" }, + "tui.select.pageUp": { defaultKeys: "pageUp", description: "Selection page up" }, + "tui.select.pageDown": { + defaultKeys: "pageDown", + description: "Selection page down", + }, + "tui.select.confirm": { defaultKeys: "enter", description: "Confirm selection" }, + "tui.select.cancel": { + defaultKeys: ["escape", "ctrl+c"], + description: "Cancel selection", + }, +} as const satisfies KeybindingDefinitions; + +export interface KeybindingConflict { + key: KeyId; + keybindings: string[]; +} const SHIFTED_SYMBOL_KEYS = new Set([ "!", @@ -116,50 +166,67 @@ const SHIFTED_SYMBOL_KEYS = new Set([ const normalizeKeyId = (key: KeyId): KeyId => key.toLowerCase() as KeyId; -/** - * Manages keybindings for the editor. - */ -export class EditorKeybindingsManager { - #actionToKeys: Map; +function normalizeKeys(keys: KeyId | KeyId[] | undefined): KeyId[] { + if (keys === undefined) return []; + const keyList = Array.isArray(keys) ? keys : [keys]; + const seen = new Set(); + const result: KeyId[] = []; + for (const key of keyList) { + const normalized = normalizeKeyId(key); + if (!seen.has(normalized)) { + seen.add(normalized); + result.push(normalized); + } + } + return result; +} - constructor(config: EditorKeybindingsConfig = {}) { - this.#actionToKeys = new Map(); - this.#buildMaps(config); +export class KeybindingsManager { + #definitions: KeybindingDefinitions; + #userBindings: KeybindingsConfig; + #keysById = new Map(); + #conflicts: KeybindingConflict[] = []; + + constructor(definitions: KeybindingDefinitions, userBindings: KeybindingsConfig = {}) { + this.#definitions = definitions; + this.#userBindings = userBindings; + this.#rebuild(); } - #buildMaps(config: EditorKeybindingsConfig): void { - this.#actionToKeys.clear(); + #rebuild(): void { + this.#keysById.clear(); + this.#conflicts = []; - // Start with defaults - for (const [action, keys] of Object.entries(DEFAULT_EDITOR_KEYBINDINGS)) { - const keyArray = Array.isArray(keys) ? keys : [keys]; - this.#actionToKeys.set( - action as EditorAction, - keyArray.map(key => normalizeKeyId(key as KeyId)), - ); + const userClaims = new Map>(); + for (const [keybinding, keys] of Object.entries(this.#userBindings)) { + if (!(keybinding in this.#definitions)) continue; + for (const key of normalizeKeys(keys)) { + const claimants = userClaims.get(key) ?? new Set(); + claimants.add(keybinding as Keybinding); + userClaims.set(key, claimants); + } } - // Override with user config - for (const [action, keys] of Object.entries(config)) { - if (keys === undefined) continue; - const keyArray = Array.isArray(keys) ? keys : [keys]; - this.#actionToKeys.set( - action as EditorAction, - keyArray.map(key => normalizeKeyId(key as KeyId)), - ); + for (const [key, keybindings] of userClaims) { + if (keybindings.size > 1) { + this.#conflicts.push({ key, keybindings: [...keybindings] }); + } + } + + for (const [id, definition] of Object.entries(this.#definitions)) { + const userKeys = this.#userBindings[id]; + const keys = userKeys === undefined ? normalizeKeys(definition.defaultKeys) : normalizeKeys(userKeys); + this.#keysById.set(id as Keybinding, keys); } } - /** - * Check if input matches a specific action. - */ - matches(data: string, action: EditorAction): boolean { - const keys = this.#actionToKeys.get(action); - if (!keys) return false; + matches(data: string, keybinding: Keybinding): boolean { + const keys = this.#keysById.get(keybinding) ?? []; for (const key of keys) { if (matchesKey(data, key)) return true; } + // Handle shifted symbol keys (e.g., shift+- produces _ on US layout) const parsed = parseKey(data); if (!parsed || !parsed.startsWith("shift+")) return false; const keyName = parsed.slice("shift+".length); @@ -167,31 +234,46 @@ export class EditorKeybindingsManager { return keys.includes(keyName as KeyId); } - /** - * Get keys bound to an action. - */ - getKeys(action: EditorAction): KeyId[] { - return this.#actionToKeys.get(action) ?? []; + getKeys(keybinding: Keybinding): KeyId[] { + return [...(this.#keysById.get(keybinding) ?? [])]; } - /** - * Update configuration. - */ - setConfig(config: EditorKeybindingsConfig): void { - this.#buildMaps(config); + getDefinition(keybinding: Keybinding): KeybindingDefinition { + return this.#definitions[keybinding]; + } + + getConflicts(): KeybindingConflict[] { + return this.#conflicts.map(conflict => ({ ...conflict, keybindings: [...conflict.keybindings] })); + } + + setUserBindings(userBindings: KeybindingsConfig): void { + this.#userBindings = userBindings; + this.#rebuild(); + } + + getUserBindings(): KeybindingsConfig { + return { ...this.#userBindings }; + } + + getResolvedBindings(): KeybindingsConfig { + const resolved: KeybindingsConfig = {}; + for (const id of Object.keys(this.#definitions)) { + const keys = this.#keysById.get(id as Keybinding) ?? []; + resolved[id] = keys.length === 1 ? keys[0]! : [...keys]; + } + return resolved; } } -// Global instance -let globalEditorKeybindings: EditorKeybindingsManager | null = null; +let globalKeybindings: KeybindingsManager | null = null; -export function getEditorKeybindings(): EditorKeybindingsManager { - if (!globalEditorKeybindings) { - globalEditorKeybindings = new EditorKeybindingsManager(); +export function setKeybindings(keybindings: KeybindingsManager): void { + globalKeybindings = keybindings; +} + +export function getKeybindings(): KeybindingsManager { + if (!globalKeybindings) { + globalKeybindings = new KeybindingsManager(TUI_KEYBINDINGS); } - return globalEditorKeybindings; -} - -export function setEditorKeybindings(manager: EditorKeybindingsManager): void { - globalEditorKeybindings = manager; + return globalKeybindings; } diff --git a/packages/tui/src/keys.ts b/packages/tui/src/keys.ts index ace1d2286..286d4e20b 100644 --- a/packages/tui/src/keys.ts +++ b/packages/tui/src/keys.ts @@ -25,6 +25,34 @@ import { parseKittySequence as parseKittySequenceNative, } from "@oh-my-pi/pi-natives"; +// ============================================================================= +// Platform Detection +// ============================================================================= + +function isWindowsTerminalSession(): boolean { + return ( + Boolean(process.env.WT_SESSION) && !process.env.SSH_CONNECTION && !process.env.SSH_CLIENT && !process.env.SSH_TTY + ); +} + +/** + * Raw 0x08 (BS) is ambiguous in legacy terminals. + * + * - Windows Terminal uses it for Ctrl+Backspace. + * - Some legacy terminals and tmux setups send it for plain Backspace. + * + * Prefer explicit Kitty / CSI-u / modifyOtherKeys sequences whenever they are + * available. Fall back to a Windows Terminal heuristic only for raw BS bytes. + */ +function matchesRawBackspace(data: string, expectedModifier: number): boolean { + if (data === "\x7f") return expectedModifier === 0; + if (data !== "\x08") return false; + // On Windows Terminal, 0x08 = Ctrl+Backspace. On others, it's plain Backspace. + return isWindowsTerminalSession() ? expectedModifier === 4 : expectedModifier === 0; +} + +export { isWindowsTerminalSession, matchesRawBackspace }; + // ============================================================================= // Global Kitty Protocol State // ============================================================================= diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 1c735b7af..7f3d39b8d 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -108,6 +108,10 @@ function parseSizeValue(value: SizeValue | undefined, referenceSize: number): nu return undefined; } +function isTermuxSession(): boolean { + return Boolean(process.env.TERMUX_VERSION); +} + /** * Options for overlay positioning and sizing. * Values can be absolute numbers or percentage strings (e.g., "50%"). @@ -204,6 +208,7 @@ export class TUI extends Container { terminal: Terminal; #previousLines: string[] = []; #previousWidth = 0; + #previousHeight = 0; #focusedComponent: Component | null = null; #inputListeners = new Set(); @@ -559,6 +564,7 @@ export class TUI extends Container { if (force) { this.#previousLines = []; this.#previousWidth = -1; // -1 triggers widthChanged, forcing a full clear + this.#previousHeight = -1; // -1 triggers heightChanged, forcing a full clear this.#cursorRow = 0; this.#hardwareCursorRow = 0; this.#viewportTopRow = 0; @@ -995,12 +1001,13 @@ export class TUI extends Container { // Width changed - need full re-render (line wrapping changes) const widthChanged = this.#previousWidth !== 0 && this.#previousWidth !== width; + const heightChanged = this.#previousHeight !== 0 && this.#previousHeight !== height; // Helper to clear scrollback and viewport and render all new lines const fullRender = (clear: boolean): void => { this.#fullRedrawCount += 1; let buffer = "\x1b[?2026h"; // Begin synchronized output - if (clear) buffer += "\x1b[3J\x1b[2J\x1b[H"; // Clear scrollback, screen, and home + if (clear) buffer += "\x1b[2J\x1b[H\x1b[3J"; // Clear screen, home, then clear scrollback const reset = SEGMENT_RESET; for (let i = 0; i < newLines.length; i++) { if (i > 0) buffer += "\r\n"; @@ -1021,6 +1028,7 @@ export class TUI extends Container { this.#positionHardwareCursor(cursorPos, newLines.length); this.#previousLines = newLines; this.#previousWidth = width; + this.#previousHeight = height; }; const debugRedraw = process.env.PI_DEBUG_REDRAW === "1"; @@ -1032,15 +1040,24 @@ export class TUI extends Container { }; // First render - just output everything without clearing (assumes clean screen) - if (this.#previousLines.length === 0 && !widthChanged) { + if (this.#previousLines.length === 0 && !widthChanged && !heightChanged) { logRedraw("first render"); fullRender(false); return; } - // Width changed - full re-render (line wrapping changes) + // Width changes always need a full re-render because wrapping changes. if (widthChanged) { - logRedraw(`width changed (${this.#previousWidth} -> ${width})`); + logRedraw(`terminal width changed (${this.#previousWidth} -> ${width})`); + fullRender(true); + return; + } + + // Height changes normally need a full re-render to keep the visible viewport aligned, + // but Termux changes height when the software keyboard shows or hides. + // In that environment, a full redraw causes the entire history to replay on every toggle. + if (heightChanged && !isTermuxSession()) { + logRedraw(`terminal height changed (${this.#previousHeight} -> ${height})`); fullRender(true); return; } @@ -1122,6 +1139,7 @@ export class TUI extends Container { this.#positionHardwareCursor(cursorPos, newLines.length); this.#previousLines = newLines; this.#previousWidth = width; + this.#previousHeight = height; this.#viewportTopRow = Math.max(0, this.#maxLinesRendered - height); return; } @@ -1270,6 +1288,7 @@ export class TUI extends Container { this.#previousLines = newLines; this.#previousWidth = width; + this.#previousHeight = height; } /** diff --git a/packages/tui/src/utils.ts b/packages/tui/src/utils.ts index df3930487..b2f38e5e5 100644 --- a/packages/tui/src/utils.ts +++ b/packages/tui/src/utils.ts @@ -35,15 +35,25 @@ export function getSegmenter(): Intl.Segmenter { /** * Calculate the visible width of a string in terminal columns. */ +function _isPrintableAscii(str: string): boolean { + for (let i = 0; i < str.length; i++) { + const code = str.charCodeAt(i); + if (code < 0x20 || code > 0x7e) { + return false; + } + } + return true; +} + export function visibleWidthRaw(str: string): number { if (!str) { return 0; } // Fast path: pure ASCII printable - let isPureAscii = true; let tabLength = 0; const tabWidth = getDefaultTabWidth(); + let isPureAscii = true; for (let i = 0; i < str.length; i++) { const code = str.charCodeAt(i); if (code === 9) { diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index c271c1f04..36f6faab2 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -3,12 +3,12 @@ import { stripVTControlCharacters } from "node:util"; import { CombinedAutocompleteProvider } from "@oh-my-pi/pi-tui/autocomplete"; import { Editor } from "@oh-my-pi/pi-tui/components/editor"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; -import { EditorKeybindingsManager, setEditorKeybindings } from "../src/keybindings"; +import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "../src/keybindings"; import { defaultEditorTheme } from "./test-themes"; describe("Editor component", () => { afterEach(() => { - setEditorKeybindings(new EditorKeybindingsManager()); + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); }); describe("Prompt history navigation", () => { @@ -1340,9 +1340,9 @@ describe("Editor component", () => { }); it("uses the configured undo binding", () => { - setEditorKeybindings( - new EditorKeybindingsManager({ - undo: "f8", + setKeybindings( + new KeybindingsManager(TUI_KEYBINDINGS, { + "tui.editor.undo": "f8", }), ); diff --git a/packages/tui/test/keybindings.test.ts b/packages/tui/test/keybindings.test.ts new file mode 100644 index 000000000..a6d616c06 --- /dev/null +++ b/packages/tui/test/keybindings.test.ts @@ -0,0 +1,37 @@ +import { describe, expect, it } from "bun:test"; +import { KeybindingsManager, TUI_KEYBINDINGS } from "@oh-my-pi/pi-tui/keybindings"; + +describe("KeybindingsManager", () => { + it("does not evict selector confirm when input submit is rebound", () => { + const keybindings = new KeybindingsManager(TUI_KEYBINDINGS, { + "tui.input.submit": ["enter", "ctrl+enter"], + }); + + expect(keybindings.getKeys("tui.input.submit")).toEqual(["enter", "ctrl+enter"]); + expect(keybindings.getKeys("tui.select.confirm")).toEqual(["enter"]); + }); + + it("does not evict cursor bindings when another action reuses the same key", () => { + const keybindings = new KeybindingsManager(TUI_KEYBINDINGS, { + "tui.select.up": ["up", "ctrl+p"], + }); + + expect(keybindings.getKeys("tui.select.up")).toEqual(["up", "ctrl+p"]); + expect(keybindings.getKeys("tui.editor.cursorUp")).toEqual(["up"]); + }); + + it("still reports direct user binding conflicts without evicting defaults", () => { + const keybindings = new KeybindingsManager(TUI_KEYBINDINGS, { + "tui.input.submit": "ctrl+x", + "tui.select.confirm": "ctrl+x", + }); + + expect(keybindings.getConflicts()).toEqual([ + { + key: "ctrl+x", + keybindings: ["tui.input.submit", "tui.select.confirm"], + }, + ]); + expect(keybindings.getKeys("tui.editor.cursorLeft")).toEqual(["left", "ctrl+b"]); + }); +}); diff --git a/packages/tui/test/markdown.test.ts b/packages/tui/test/markdown.test.ts index 58dc63616..86d89c541 100644 --- a/packages/tui/test/markdown.test.ts +++ b/packages/tui/test/markdown.test.ts @@ -471,6 +471,22 @@ describe("Markdown component", () => { const tableRow = plainLines.find(line => line.includes("|")); expect(tableRow?.startsWith(" "), "Table should have left padding").toBeTruthy(); }); + + it("should not add a trailing blank line when table is the last rendered block", () => { + const markdown = new Markdown( + `| Name | +| --- | +| Alice |`, + 0, + 0, + defaultMarkdownTheme, + ); + + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines.at(-1)).not.toBe(""); + }); }); describe("Combined features", () => { @@ -624,6 +640,44 @@ again, hello world`, `Expected 1 empty line after code block, but found ${emptyLineCount}. Lines after backticks: ${JSON.stringify(afterBackticks.slice(0, 5))}`, ).toBe(1); }); + + it("should normalize paragraph and code block spacing to one blank line", () => { + const cases = [ + `hello this is text +\`\`\` +code block +\`\`\` +more text`, + `hello this is text + +\`\`\` +code block +\`\`\` + +more text`, + ]; + const expectedLines = ["hello this is text", "", "```", " code block", "```", "", "more text"]; + + for (const text of cases) { + const markdown = new Markdown(text, 0, 0, defaultMarkdownTheme); + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines).toEqual(expectedLines); + } + }); + + it("should not add a trailing blank line when code block is the last rendered block", () => { + const cases = ["```js\nconst hello = 'world';\n```", "hello world\n\n```js\nconst hello = 'world';\n```"]; + + for (const text of cases) { + const markdown = new Markdown(text, 0, 0, defaultMarkdownTheme); + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines.at(-1)).not.toBe(""); + } + }); }); describe("Spacing after dividers", () => { @@ -653,6 +707,14 @@ again, hello world`, `Expected 1 empty line after divider, but found ${emptyLineCount}. Lines after divider: ${JSON.stringify(afterDivider.slice(0, 5))}`, ).toBe(1); }); + + it("should not add a trailing blank line when divider is the last rendered block", () => { + const markdown = new Markdown("---", 0, 0, defaultMarkdownTheme); + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines.at(-1)).not.toBe(""); + }); }); describe("Spacing after headings", () => { @@ -680,6 +742,14 @@ This is a paragraph`, `Expected 1 empty line after heading, but found ${emptyLineCount}. Lines after heading: ${JSON.stringify(afterHeading.slice(0, 5))}`, ).toBe(1); }); + + it("should not add a trailing blank line when heading is the last rendered block", () => { + const markdown = new Markdown("# Hello", 0, 0, defaultMarkdownTheme); + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines.at(-1)).not.toBe(""); + }); }); describe("Spacing after blockquotes", () => { @@ -709,6 +779,14 @@ again, hello world`, `Expected 1 empty line after blockquote, but found ${emptyLineCount}. Lines after quote: ${JSON.stringify(afterQuote.slice(0, 5))}`, ).toBe(1); }); + + it("should not add a trailing blank line when blockquote is the last rendered block", () => { + const markdown = new Markdown("> This is a quote", 0, 0, defaultMarkdownTheme); + const lines = markdown.render(80); + const plainLines = lines.map(line => line.replace(/\x1b\[[0-9;]*m/g, "").trimEnd()); + + expect(plainLines.at(-1)).not.toBe(""); + }); }); describe("Blockquotes with multiline content", () => { diff --git a/packages/tui/test/select-list.test.ts b/packages/tui/test/select-list.test.ts new file mode 100644 index 000000000..f9a92765c --- /dev/null +++ b/packages/tui/test/select-list.test.ts @@ -0,0 +1,171 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; +import { SelectList } from "../src/components/select-list"; +import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "../src/keybindings"; + +const testTheme = { + selectedPrefix: (text: string) => text, + selectedText: (text: string) => text, + description: (text: string) => text, + scrollInfo: (text: string) => text, + noMatch: (text: string) => text, + symbols: { + cursor: "→", + inputCursor: "|", + hrChar: "─", + quoteBorder: "│", + boxRound: { topLeft: "╭", topRight: "╮", bottomLeft: "╰", bottomRight: "╯", horizontal: "─", vertical: "│" }, + boxSharp: { + topLeft: "┌", + topRight: "┐", + bottomLeft: "└", + bottomRight: "┘", + horizontal: "─", + vertical: "│", + teeDown: "┬", + teeUp: "┴", + teeLeft: "┤", + teeRight: "├", + cross: "┼", + }, + table: { + topLeft: "┌", + topRight: "┐", + bottomLeft: "└", + bottomRight: "┘", + horizontal: "─", + vertical: "│", + teeDown: "┬", + teeUp: "┴", + teeLeft: "┤", + teeRight: "├", + cross: "┼", + }, + spinnerFrames: ["|"], + }, +}; + +const visibleIndexOf = (line: string, text: string): number => { + const index = line.indexOf(text); + expect(index).not.toBe(-1); + return visibleWidth(line.slice(0, index)); +}; + +describe("SelectList", () => { + beforeEach(() => { + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); + }); + + afterEach(() => { + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); + }); + + it("normalizes multiline descriptions to single line", () => { + const items = [ + { + value: "test", + label: "test", + description: "Line one\nLine two\nLine three", + }, + ]; + + const list = new SelectList(items, 5, testTheme); + const rendered = list.render(80); + + expect(rendered.length).toBeGreaterThanOrEqual(1); + expect(rendered[0]).not.toContain("\n"); + expect(rendered[0]).toContain("Line one Line two Line three"); + }); + + it("keeps descriptions aligned when the primary text is truncated", () => { + const items = [ + { value: "short", label: "short", description: "short description" }, + { + value: "very-long-command-name-that-needs-truncation", + label: "very-long-command-name-that-needs-truncation", + description: "long description", + }, + ]; + + const list = new SelectList(items, 5, testTheme); + const rendered = list.render(80); + + expect(visibleIndexOf(rendered[0], "short description")).toBe(visibleIndexOf(rendered[1], "long description")); + }); + + it("uses the configured minimum primary column width", () => { + const items = [ + { value: "a", label: "a", description: "first" }, + { value: "bb", label: "bb", description: "second" }, + ]; + + const list = new SelectList(items, 5, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 20, + }); + const rendered = list.render(80); + + expect(rendered[0].indexOf("first")).toBe(14); + expect(rendered[1].indexOf("second")).toBe(14); + }); + + it("uses the configured maximum primary column width", () => { + const items = [ + { + value: "very-long-command-name-that-needs-truncation", + label: "very-long-command-name-that-needs-truncation", + description: "first", + }, + { value: "short", label: "short", description: "second" }, + ]; + + const list = new SelectList(items, 5, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 20, + }); + const rendered = list.render(80); + + expect(visibleIndexOf(rendered[0], "first")).toBe(22); + expect(visibleIndexOf(rendered[1], "second")).toBe(22); + }); + + it("allows overriding primary truncation while preserving description alignment", () => { + const items = [ + { + value: "very-long-command-name-that-needs-truncation", + label: "very-long-command-name-that-needs-truncation", + description: "first", + }, + { value: "short", label: "short", description: "second" }, + ]; + + const list = new SelectList(items, 5, testTheme, { + minPrimaryColumnWidth: 12, + maxPrimaryColumnWidth: 12, + truncatePrimary: ({ text, maxWidth }) => { + if (text.length <= maxWidth) { + return text; + } + + return `${text.slice(0, Math.max(0, maxWidth - 1))}…`; + }, + }); + const rendered = list.render(80); + + expect(rendered[0]).toContain("…"); + expect(visibleIndexOf(rendered[0], "first")).toBe(visibleIndexOf(rendered[1], "second")); + }); + + it("confirms the selected item when Enter arrives as LF", () => { + const items = [{ value: "run", label: "run" }]; + const list = new SelectList(items, 5, testTheme); + let selectedValue: string | undefined; + list.onSelect = item => { + selectedValue = item.value; + }; + + list.handleInput("\n"); + + expect(selectedValue).toBe("run"); + }); +}); diff --git a/packages/tui/test/settings-list.test.ts b/packages/tui/test/settings-list.test.ts new file mode 100644 index 000000000..ad7d87134 --- /dev/null +++ b/packages/tui/test/settings-list.test.ts @@ -0,0 +1,47 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { SettingsList, type SettingsListTheme } from "../src/components/settings-list"; +import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "../src/keybindings"; + +const testTheme: SettingsListTheme = { + label: (text: string) => text, + value: (text: string) => text, + description: (text: string) => text, + cursor: "→ ", + hint: (text: string) => text, +}; + +describe("SettingsList", () => { + beforeEach(() => { + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); + }); + + afterEach(() => { + setKeybindings(new KeybindingsManager(TUI_KEYBINDINGS)); + }); + + it("cycles the selected value when Enter arrives as LF", () => { + const changes: Array<[string, string]> = []; + const list = new SettingsList( + [ + { + id: "mode", + label: "Mode", + currentValue: "off", + values: ["off", "on"], + }, + ], + 5, + testTheme, + (id, value) => { + changes.push([id, value]); + }, + () => { + throw new Error("cancel should not be called"); + }, + ); + + list.handleInput("\n"); + + expect(changes).toEqual([["mode", "on"]]); + }); +}); diff --git a/packages/tui/test/truncate-to-width.test.ts b/packages/tui/test/truncate-to-width.test.ts new file mode 100644 index 000000000..8d879f358 --- /dev/null +++ b/packages/tui/test/truncate-to-width.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from "bun:test"; +import { Ellipsis, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui/utils"; + +describe("truncateToWidth", () => { + it("keeps output within width for very large unicode input", () => { + const text = "🙂界".repeat(100_000); + const truncated = truncateToWidth(text, 40, Ellipsis.Unicode); + + expect(visibleWidth(truncated)).toBeLessThanOrEqual(40); + }); + + it("preserves ANSI styling for kept text", () => { + const text = `\x1b[31m${"hello ".repeat(1000)}\x1b[0m`; + const truncated = truncateToWidth(text, 20, Ellipsis.Unicode); + + expect(visibleWidth(truncated)).toBeLessThanOrEqual(20); + expect(truncated.includes("\x1b[31m")).toBe(true); + }); + + it("handles malformed ANSI escape prefixes without hanging", () => { + const text = `abc\x1bnot-ansi ${"🙂".repeat(1000)}`; + // Should complete without hanging — the exact width depends on how the + // native implementation classifies the malformed escape prefix. + const truncated = truncateToWidth(text, 20, Ellipsis.Unicode); + expect(typeof truncated).toBe("string"); + }); + + it("returns the original text when it already fits", () => { + expect(truncateToWidth("a", 2, Ellipsis.Unicode)).toBe("a"); + expect(truncateToWidth("界", 2, Ellipsis.Unicode)).toBe("界"); + }); + + it("pads truncated output to requested width", () => { + const truncated = truncateToWidth("🙂界🙂界🙂界", 8, Ellipsis.Unicode, true); + expect(visibleWidth(truncated)).toBe(8); + }); + + it("adds a trailing reset when truncating without an ellipsis", () => { + const truncated = truncateToWidth(`\x1b[31m${"hello".repeat(100)}`, 10, Ellipsis.Omit); + expect(visibleWidth(truncated)).toBeLessThanOrEqual(10); + expect(truncated.endsWith("\x1b[0m")).toBe(true); + }); +}); + +describe("visibleWidth", () => { + it("counts tabs inline and skips ANSI inline", () => { + expect(visibleWidth("\t\x1b[31m界\x1b[0m")).toBe(5); + }); +}); From 7e56323c63c2df22982623c6f2f7de3d21c3cb75 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:12:23 +0100 Subject: [PATCH 09/22] fix(ai): corrected lazy stream result forwarding and error handling - Fixed lazy stream forwarding to properly handle final results from source streams with `result()` methods. - Fixed lazy stream error handling to convert iterator failures into terminal error results instead of silently failing. - Added comprehensive test coverage for lazy stream result forwarding and error handling scenarios. --- packages/ai/CHANGELOG.md | 3 + .../ai/src/providers/register-builtins.ts | 30 +++++- packages/ai/test/register-builtins.test.ts | 95 +++++++++++++++++++ 3 files changed, 123 insertions(+), 5 deletions(-) create mode 100644 packages/ai/test/register-builtins.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index c3c83f93b..531a6324e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,12 +1,15 @@ # Changelog ## [Unreleased] + ### Added - Added `isUsageLimitError()` to `rate-limit-utils` as a single source of truth for detecting usage/quota limit errors across all providers ### Fixed +- Fixed lazy stream forwarding to properly handle final results from source streams with `result()` methods +- Fixed lazy stream error handling to convert iterator failures into terminal error results instead of silently failing - Fixed `parseRateLimitReason` to recognize "usage limit" in error messages and correctly classify them as `QUOTA_EXHAUSTED` - Fixed Codex `fetchWithRetry` retrying 429 responses for `usage_limit_reached` errors for up to 5 minutes instead of returning immediately for credential switching - Removed `usage.?limit` from `TRANSIENT_MESSAGE_PATTERN` in retry utils since usage limits are not transient and require credential rotation diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts index ab7ba9eef..dc89cbd30 100644 --- a/packages/ai/src/providers/register-builtins.ts +++ b/packages/ai/src/providers/register-builtins.ts @@ -147,12 +147,32 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void { // Stream forwarding / error helpers // --------------------------------------------------------------------------- -function forwardStream(target: EventStreamImpl, source: AsyncIterable): void { +function hasFinalResult( + source: AsyncIterable, +): source is AsyncIterable & { result(): Promise } { + return typeof (source as { result?: unknown }).result === "function"; +} + +function forwardStream( + target: EventStreamImpl, + source: AsyncIterable, + model: Model, +): void { (async () => { - for await (const event of source) { - target.push(event); + try { + for await (const event of source) { + target.push(event); + } + if (hasFinalResult(source)) { + target.end(await source.result()); + } else { + target.end(); + } + } catch (error) { + const message = createLazyLoadErrorMessage(model, error); + target.push({ type: "error", reason: "error", error: message }); + target.end(message); } - target.end(); })(); } @@ -190,7 +210,7 @@ function createLazyStream( loadModule() .then(module => { const inner = module.stream(model, context, options); - forwardStream(outer, inner); + forwardStream(outer, inner, model); }) .catch(error => { const message = createLazyLoadErrorMessage(model, error); diff --git a/packages/ai/test/register-builtins.test.ts b/packages/ai/test/register-builtins.test.ts new file mode 100644 index 000000000..f3911edcb --- /dev/null +++ b/packages/ai/test/register-builtins.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from "bun:test"; +import { setBedrockProviderModule, streamBedrock } from "../src/providers/register-builtins"; +import type { AssistantMessage, Context, Model } from "../src/types"; +import type { AssistantMessageEventStream } from "../src/utils/event-stream"; + +function createModel(): Model<"bedrock-converse-stream"> { + return { + id: "mock-bedrock", + name: "Mock Bedrock", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + baseUrl: "https://example.invalid", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8192, + maxTokens: 2048, + }; +} + +function createAssistantMessage( + stopReason: AssistantMessage["stopReason"] = "stop", + errorMessage?: string, +): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text: errorMessage ? `error: ${errorMessage}` : "ok" }], + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + model: "mock-bedrock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason, + errorMessage, + timestamp: Date.now(), + }; +} + +const baseContext: Context = { messages: [] }; + +describe("register-builtins lazy streams", () => { + it("resolves the outer stream result from source.result() when no terminal event is iterated", async () => { + const finalMessage = createAssistantMessage("stop"); + const partialMessage = createAssistantMessage("stop"); + const source = { + async *[Symbol.asyncIterator]() { + yield { type: "start", partial: partialMessage } as const; + }, + result: async () => finalMessage, + } as unknown as AssistantMessageEventStream; + + setBedrockProviderModule({ + streamBedrock: () => source, + }); + + const stream = streamBedrock(createModel(), baseContext, {}); + const result = await Promise.race([stream.result(), Bun.sleep(100).then(() => "timeout" as const)]); + + expect(result).not.toBe("timeout"); + if (result === "timeout") { + throw new Error("Timed out waiting for forwarded stream result"); + } + expect(result).toEqual(finalMessage); + }); + + it("turns iterator failures into terminal error results", async () => { + const partialMessage = createAssistantMessage("stop"); + const source = { + async *[Symbol.asyncIterator]() { + yield { type: "start", partial: partialMessage } as const; + throw new Error("bedrock exploded"); + }, + } as unknown as AssistantMessageEventStream; + + setBedrockProviderModule({ + streamBedrock: () => source, + }); + + const stream = streamBedrock(createModel(), baseContext, {}); + const result = await Promise.race([stream.result(), Bun.sleep(100).then(() => "timeout" as const)]); + + expect(result).not.toBe("timeout"); + if (result === "timeout") { + throw new Error("Timed out waiting for forwarded error result"); + } + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("bedrock exploded"); + }); +}); From 7d3ff27c64c87e442dc4aae0a0a4947e03c40407 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:12:55 +0100 Subject: [PATCH 10/22] fix(editor): corrected editor consuming rebound copy keys preventing custom bindings - Fixed editor consuming user-rebound copy keys, preventing custom keybindings from working. - Changed copy key detection from generic `tui.input.copy` binding to explicit `ctrl+c` check to avoid swallowing user-rebound keys. - Added test case verifying editor does not consume keys rebound to copy action. --- packages/tui/CHANGELOG.md | 5 +++++ packages/tui/src/components/editor.ts | 6 ++++-- packages/tui/test/editor.test.ts | 17 +++++++++++++++++ 3 files changed, 26 insertions(+), 2 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index d13917853..2f717af51 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -1,10 +1,15 @@ # Changelog ## [Unreleased] + ### Added - Added `renderInlineMarkdown()` function to render inline markdown (bold, italic, code, links, strikethrough) to styled strings +### Fixed + +- Fixed editor consuming user-rebound copy keys, preventing custom keybindings from working in the editor + ## [13.14.1] - 2026-03-21 ### Added diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 3095d6f3a..93d4c32d9 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -732,8 +732,10 @@ export class Editor implements Component, Focusable { // Handle special key combinations first - // Ctrl+C - Exit (let parent handle this) - if (kb.matches(data, "tui.input.copy")) { + // Ctrl+C is reserved by parent components for app-level handling. + // Do not consume arbitrary user-bound "copy" keys here, since the editor + // has no copy implementation and would make those keys disappear. + if (matchesKey(data, "ctrl+c")) { return; } diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index 36f6faab2..dfa9534bc 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -1356,6 +1356,23 @@ describe("Editor component", () => { expect(editor.getCursor()).toEqual({ line: 0, col: 0 }); }); + it("does not swallow keys rebound to copy", () => { + setKeybindings( + new KeybindingsManager(TUI_KEYBINDINGS, { + "tui.input.copy": "left", + }), + ); + + const editor = new Editor(defaultEditorTheme); + editor.setText("ab"); + + editor.handleInput("\x1b[D"); // Left arrow + editor.handleInput("X"); + + expect(editor.getText()).toBe("aXb"); + expect(editor.getCursor()).toEqual({ line: 0, col: 2 }); + }); + it("undoes the last paste when a transient #undo trigger is executed", () => { const editor = new Editor(defaultEditorTheme); From 5d2e2cef1f1e92ade9255cfc0e5937750e783b73 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:12:45 +0100 Subject: [PATCH 11/22] feat(react-edit-benchmark): added retry mechanism and autocorrect tracking to benchmarks - Added retry mechanism for benchmark tasks with separate system and retry prompt templates to improve edit success rates. - Introduced autocorrect tracking metrics including autocorrect-free success rate and edit autocorrect counts in task and benchmark summaries. - Refactored prompt building into modular functions (buildBenchmarkSystemPrompt, buildInitialBenchmarkPrompt, buildRetryBenchmarkPrompt) with BenchmarkPromptDelivery type for distinguishing initial and follow-up messages. - Added session management with cache-keyed provider session IDs using xxHash64 and centralized RPC argument building via prepareBenchmarkSessionSetup. --- packages/coding-agent/src/cli/args.ts | 3 + packages/coding-agent/src/main.ts | 3 + packages/coding-agent/src/sdk.ts | 31 +- packages/coding-agent/test/args.test.ts | 5 + .../src/prompts/benchmark-retry.md | 12 + .../src/prompts/benchmark-system.md | 20 ++ .../src/prompts/benchmark-task.md | 21 -- packages/react-edit-benchmark/src/report.ts | 7 + packages/react-edit-benchmark/src/runner.ts | 266 +++++++++++++++--- 9 files changed, 303 insertions(+), 65 deletions(-) create mode 100644 packages/react-edit-benchmark/src/prompts/benchmark-retry.md create mode 100644 packages/react-edit-benchmark/src/prompts/benchmark-system.md diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 751dd271b..a7adea1e7 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -28,6 +28,7 @@ export interface Args { mode?: Mode; noSession?: boolean; sessionDir?: string; + providerSessionId?: string; fork?: string; models?: string[]; tools?: string[]; @@ -98,6 +99,8 @@ export function parseArgs(args: string[], extensionFlags?: Map --model diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index c2413c737..c2cf91860 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -144,6 +144,9 @@ export interface CreateAgentSessionOptions { /** System prompt. String replaces default, function receives default and returns final. */ systemPrompt?: string | ((defaultPrompt: string) => string); + /** Optional provider-facing session identifier for prompt caches and sticky auth selection. + * Keeps persisted session files isolated while reusing provider-side caches. */ + providerSessionId?: string; /** Custom tools to register (in addition to built-in tools). Accepts both CustomTool and ToolDefinition. */ customTools?: (CustomTool | ToolDefinition)[]; @@ -667,7 +670,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} logger.time("sessionManager", () => SessionManager.create(cwd, SessionManager.getDefaultSessionDir(cwd, agentDir)), ); - const sessionId = sessionManager.getSessionId(); + const providerSessionId = options.providerSessionId ?? sessionManager.getSessionId(); const modelApiKeyAvailability = new Map(); const getModelAvailabilityKey = (candidate: Model): string => `${candidate.provider}\u0000${candidate.baseUrl ?? ""}`; @@ -678,7 +681,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return cached; } - const hasKey = !!(await modelRegistry.getApiKey(candidate, sessionId)); + const hasKey = !!(await modelRegistry.getApiKey(candidate, providerSessionId)); modelApiKeyAvailability.set(availabilityKey, hasKey); return hasKey; }; @@ -1285,9 +1288,15 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const normalizedRequested = requestedToolNames.filter(name => toolRegistry.has(name)); const includeExitPlanMode = requestedToolNames.includes("exit_plan_mode"); const mcpDiscoveryEnabled = settings.get("mcp.discoveryMode") ?? false; + const defaultInactiveToolNames = new Set( + registeredTools.filter(tool => tool.definition.defaultInactive).map(tool => tool.definition.name), + ); const requestedActiveToolNames = includeExitPlanMode ? normalizedRequested : normalizedRequested.filter(name => name !== "exit_plan_mode"); + const initialRequestedActiveToolNames = options.toolNames + ? requestedActiveToolNames + : requestedActiveToolNames.filter(name => !defaultInactiveToolNames.has(name)); const explicitlyRequestedMCPToolNames = options.toolNames ? requestedActiveToolNames.filter(name => name.startsWith("mcp_")) : []; @@ -1302,7 +1311,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} : []; let initialSelectedMCPToolNames: string[] = []; let defaultSelectedMCPToolNames: string[] = []; - let initialToolNames = [...requestedActiveToolNames]; + let initialToolNames = [...initialRequestedActiveToolNames]; if (mcpDiscoveryEnabled) { const restoredSelectedMCPToolNames = existingSession.selectedMCPToolNames.filter(name => toolRegistry.has(name)); defaultSelectedMCPToolNames = [ @@ -1313,7 +1322,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} : [...new Set([...restoredSelectedMCPToolNames, ...defaultSelectedMCPToolNames])]; initialToolNames = [ ...new Set([ - ...requestedActiveToolNames.filter(name => !name.startsWith("mcp_")), + ...initialRequestedActiveToolNames.filter(name => !name.startsWith("mcp_")), ...initialSelectedMCPToolNames, ]), ]; @@ -1322,7 +1331,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Custom tools and extension-registered tools are always included regardless of toolNames filter const alwaysInclude: string[] = [ ...(options.customTools?.map(t => (isCustomTool(t) ? t.name : t.name)) ?? []), - ...registeredTools.map(t => t.definition.name), + ...registeredTools.filter(t => !t.definition.defaultInactive).map(t => t.definition.name), ]; for (const name of alwaysInclude) { if (mcpDiscoveryEnabled && name.startsWith("mcp_")) { @@ -1428,7 +1437,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }, convertToLlm: convertToLlmFinal, onPayload, - sessionId: sessionManager.getSessionId(), + sessionId: providerSessionId, transformContext, steeringMode: settings.get("steeringMode") ?? "one-at-a-time", followUpMode: settings.get("followUpMode") ?? "one-at-a-time", @@ -1445,9 +1454,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} preferWebsockets: preferOpenAICodexWebsockets, getToolContext: tc => toolContextStore.getContext(tc), getApiKey: async provider => { - // Use the provider argument from the in-flight request; - // agent.state.model may already be switched mid-turn. - const key = await modelRegistry.getApiKeyForProvider(provider, sessionId); + // Use the provider-facing session id for sticky credential selection so cache keys + // and provider auth affinity stay aligned across fresh benchmark sessions. + const key = await modelRegistry.getApiKeyForProvider(provider, providerSessionId); if (!key) { throw new Error(`No API key found for provider "${provider}"`); } @@ -1521,8 +1530,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} if (model?.api === "openai-codex-responses") { try { await logger.timeAsync("prewarmCodexWebsocket", prewarmOpenAICodexResponses, model, { - apiKey: await modelRegistry.getApiKey(model, sessionId), - sessionId, + apiKey: await modelRegistry.getApiKey(model, providerSessionId), + sessionId: providerSessionId, preferWebsockets: preferOpenAICodexWebsockets, providerSessionState: session.providerSessionState, }); diff --git a/packages/coding-agent/test/args.test.ts b/packages/coding-agent/test/args.test.ts index 0ebea8143..2eca63e24 100644 --- a/packages/coding-agent/test/args.test.ts +++ b/packages/coding-agent/test/args.test.ts @@ -119,6 +119,11 @@ describe("parseArgs", () => { expect(result.appendSystemPrompt).toBe("Additional context"); }); + test("parses --provider-session-id", () => { + const result = parseArgs(["--provider-session-id", "reb_cache_key"]); + expect(result.providerSessionId).toBe("reb_cache_key"); + }); + test("parses --mode", () => { const result = parseArgs(["--mode", "json"]); expect(result.mode).toBe("json"); diff --git a/packages/react-edit-benchmark/src/prompts/benchmark-retry.md b/packages/react-edit-benchmark/src/prompts/benchmark-retry.md new file mode 100644 index 000000000..95c16d1bb --- /dev/null +++ b/packages/react-edit-benchmark/src/prompts/benchmark-retry.md @@ -0,0 +1,12 @@ +Additional context for the same benchmark task. + +{{#if guided_context}} +## Guided fix (authoritative) + +{{guided_context}} +{{/if}} +## Retry context + +{{retry_context}} + +Apply one minimal concrete edit attempt using this new information. diff --git a/packages/react-edit-benchmark/src/prompts/benchmark-system.md b/packages/react-edit-benchmark/src/prompts/benchmark-system.md new file mode 100644 index 000000000..992e4487c --- /dev/null +++ b/packages/react-edit-benchmark/src/prompts/benchmark-system.md @@ -0,0 +1,20 @@ +You are participating in a code-edit benchmark inside a repository with {{#if multiFile}}multiple unrelated files{{else}}a single edit task{{/if}}. + +This benchmark is scored on exactness. Get the edit right. + +## Important constraints +- Make the minimum change necessary. Do not refactor, improve, or clean up other code. +- If you see multiple similar patterns, only change the ONE that is buggy (there is only one intended mutation). +- Preserve exact code structure. Do not rearrange statements or change formatting. +- Your output is verified by exact text diff against an expected fixture. Equivalent code, reordered imports, reordered object keys, or formatting changes will fail. +- Prefer copying the original line(s) and changing only the specific token(s) required. Do not rewrite whole statements. +- Never modify comments or license headers unless the task explicitly asks. +- Re-read the changed region after editing to confirm you only touched the intended line(s). +{{#if multiFile}}- Only modify the file(s) referenced by the task or follow-up messages. Leave all other files unchanged. +{{/if}} +## Process +- Treat the first user message as the task definition. +- Treat later follow-up messages as incremental retry context for the same task. +- Use follow-up guidance to correct the previous attempt without forgetting the original task. + +{{instructions}} diff --git a/packages/react-edit-benchmark/src/prompts/benchmark-task.md b/packages/react-edit-benchmark/src/prompts/benchmark-task.md index effa48cd2..c8a67e9fc 100644 --- a/packages/react-edit-benchmark/src/prompts/benchmark-task.md +++ b/packages/react-edit-benchmark/src/prompts/benchmark-task.md @@ -1,5 +1,3 @@ -You are working in a repository with {{#if multiFile}}multiple unrelated files{{else}}a single edit task{{/if}}. - {{task_prompt}} {{#if guided_context}} @@ -7,22 +5,3 @@ You are working in a repository with {{#if multiFile}}multiple unrelated files{{ {{guided_context}} {{/if}} - -{{#if retry_context}} -## Retry context - -{{retry_context}} -{{/if}} - -## Important constraints -- Make the minimum change necessary. Do not refactor, improve, or "clean up" other code. -- If you see multiple similar patterns, only change the ONE that is buggy (there is only one intended mutation). -- Preserve exact code structure. Do not rearrange statements or change formatting. -- Your output is verified by exact text diff against an expected fixture. “Equivalent” code, reordered imports, reordered object keys, or formatting changes will fail. -- Prefer copying the original line(s) and changing only the specific token(s) required. Do not rewrite whole statements. -- Never modify comments/license headers unless the task explicitly asks. -- After applying the fix, re-read the changed region to confirm you only touched the intended line(s). -{{#if multiFile}}- Only modify the file(s) referenced by this request. Leave all other files unchanged. -{{/if}} - -{{instructions}} diff --git a/packages/react-edit-benchmark/src/report.ts b/packages/react-edit-benchmark/src/report.ts index 56bee4c7e..78256cb5a 100644 --- a/packages/react-edit-benchmark/src/report.ts +++ b/packages/react-edit-benchmark/src/report.ts @@ -103,6 +103,13 @@ export function generateReport(result: BenchmarkResult): string { lines.push(`| Total Runs | ${summary.totalRuns} |`); lines.push(`| Successful Runs | ${summary.successfulRuns} |`); lines.push(`| **Task Success Rate** | **${formatRate(successRuns, summary.totalRuns)}** |`); + if (config.editVariant === "hashline") { + lines.push( + `| **Autocorrect-Free Success Rate** | **${formatRate(summary.autocorrectFreeSuccessfulRuns, summary.totalRuns)}** |`, + ); + lines.push(`| Autocorrected Runs | ${formatRate(summary.autocorrectedRuns, summary.totalRuns)} |`); + lines.push(`| Edit Autocorrect Rate | ${formatPercent(summary.editAutocorrectRate)} |`); + } lines.push(`| Verified Rate | ${formatRate(verifiedRuns, summary.totalRuns)} |`); lines.push(`| Edit Tool Usage Rate | ${formatRate(editToolRuns, summary.totalRuns)} |`); lines.push(`| **Edit Success Rate** | **${formatPercent(summary.editSuccessRate)}** |`); diff --git a/packages/react-edit-benchmark/src/runner.ts b/packages/react-edit-benchmark/src/runner.ts index d7fd303a3..d16d0b560 100644 --- a/packages/react-edit-benchmark/src/runner.ts +++ b/packages/react-edit-benchmark/src/runner.ts @@ -13,6 +13,8 @@ import { computeLineHash, RpcClient, renderPromptTemplate } from "@oh-my-pi/pi-c import { Snowflake } from "@oh-my-pi/pi-utils"; import { diffLines } from "diff"; import { formatDirectory } from "./formatter"; +import benchmarkRetryPrompt from "./prompts/benchmark-retry.md" with { type: "text" }; +import benchmarkSystemPrompt from "./prompts/benchmark-system.md" with { type: "text" }; import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" }; import type { EditTask } from "./tasks"; import { verifyExpectedFileSubset, verifyExpectedFiles } from "./verify"; @@ -461,20 +463,108 @@ function buildInstructions(config: BenchmarkConfig): string { : "Read the relevant files first, then use the edit tool to apply the fix."; } -function buildBenchmarkPrompt(params: { - multiFile: boolean; +type BenchmarkPromptDelivery = { + kind: "prompt" | "followUp"; + message: string; +}; + +function buildBenchmarkSystemPrompt(params: { multiFile: boolean; config: BenchmarkConfig }): string { + return renderPromptTemplate(benchmarkSystemPrompt, { + multiFile: params.multiFile, + instructions: buildInstructions(params.config), + }); +} + +function buildInitialBenchmarkPrompt(params: { taskPrompt: string; guidedContext?: string | null }): string { + return renderPromptTemplate(benchmarkTaskPrompt, { + task_prompt: params.taskPrompt, + guided_context: params.guidedContext ?? undefined, + }); +} + +function buildRetryBenchmarkPrompt(params: { retryContext: string; guidedContext?: string | null }): string { + return renderPromptTemplate(benchmarkRetryPrompt, { + retry_context: params.retryContext, + guided_context: params.guidedContext ?? undefined, + }); +} + +function buildBenchmarkPromptDelivery(params: { taskPrompt: string; guidedContext?: string | null; retryContext?: string | null; +}): BenchmarkPromptDelivery { + if (params.retryContext) { + return { + kind: "followUp", + message: buildRetryBenchmarkPrompt({ + retryContext: params.retryContext, + guidedContext: params.guidedContext, + }), + }; + } + + return { + kind: "prompt", + message: buildInitialBenchmarkPrompt({ + taskPrompt: params.taskPrompt, + guidedContext: params.guidedContext, + }), + }; +} + +const BENCHMARK_PROVIDER_SESSION_VERSION = 1; + +function buildBenchmarkProviderSessionId(params: { config: BenchmarkConfig; + task: EditTask; + multiFile: boolean; + initialGuidedContext?: string | null; }): string { - return renderPromptTemplate(benchmarkTaskPrompt, { + const keyMaterial = [ + `version:${BENCHMARK_PROVIDER_SESSION_VERSION}`, + `provider:${params.config.provider}`, + `model:${params.config.model}`, + `task:${params.task.id}`, + `system:${buildBenchmarkSystemPrompt({ multiFile: params.multiFile, config: params.config })}`, + `initial:${buildInitialBenchmarkPrompt({ taskPrompt: params.task.prompt, guidedContext: params.initialGuidedContext })}`, + ].join("\n"); + return `reb_${Bun.hash.xxHash64(keyMaterial).toString(36)}`; +} + +async function prepareBenchmarkSessionSetup(params: { + config: BenchmarkConfig; + task: EditTask; + cwd: string; + expectedDir: string; + multiFile: boolean; +}): Promise<{ initialGuidedContext: string | null; providerSessionId: string; rpcArgs: string[] }> { + const initialGuidedContext = await buildGuidedContext(params.task, params.cwd, params.expectedDir, params.config); + const providerSessionId = buildBenchmarkProviderSessionId({ + config: params.config, + task: params.task, multiFile: params.multiFile, - task_prompt: params.taskPrompt, - guided_context: params.guidedContext ?? undefined, - retry_context: params.retryContext ?? undefined, - instructions: buildInstructions(params.config), + initialGuidedContext, }); + return { + initialGuidedContext, + providerSessionId, + rpcArgs: buildBenchmarkRpcArgs(params.config, params.multiFile, providerSessionId), + }; +} + +function buildBenchmarkRpcArgs(config: BenchmarkConfig, multiFile: boolean, providerSessionId: string): string[] { + return [ + "--provider-session-id", + providerSessionId, + "--append-system-prompt", + buildBenchmarkSystemPrompt({ multiFile, config }), + "--tools", + "read,edit,write", + "--no-skills", + "--no-title", + "--no-rules", + ]; } export interface TokenStats { @@ -489,6 +579,8 @@ export interface ToolCallStats { write: number; editSuccesses: number; editFailures: number; + editWarnings: number; + editAutocorrects: number; totalInputChars: number; } @@ -517,6 +609,8 @@ export interface TaskRunResult { diff?: string; toolCalls: ToolCallStats; editFailures: EditFailure[]; + editWarnings: string[]; + editAutocorrectCount: number; /** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */ hashlineEditSubtypes?: Record; mutationIntentMatched?: boolean; @@ -542,6 +636,7 @@ export interface TaskResult { avgIndentScore: number; avgToolCalls: ToolCallStats; editSuccessRate: number; + autocorrectFreeSuccessRate: number; } export interface BenchmarkSummary { @@ -559,6 +654,10 @@ export interface BenchmarkSummary { totalToolCalls: ToolCallStats; avgToolCallsPerRun: ToolCallStats; editSuccessRate: number; + autocorrectFreeSuccessfulRuns: number; + autocorrectFreeSuccessRate: number; + autocorrectedRuns: number; + editAutocorrectRate: number; timeoutRuns: number; mutationIntentMatchRate?: number; /** Hashline edit subtype totals — only when editVariant is hashline */ @@ -611,6 +710,8 @@ async function runSingleTask( let agentResponse: string | undefined; let diff: string | undefined; const editFailures: EditFailure[] = []; + const editWarnings: string[] = []; + let editAutocorrectCount = 0; let timeoutTelemetry: PromptAttemptTelemetry | undefined; let mutationIntentValidation: MutationIntentValidation | null = null; const toolStats = { @@ -619,6 +720,8 @@ async function runSingleTask( write: 0, editSuccesses: 0, editFailures: 0, + editWarnings: 0, + editAutocorrects: 0, totalInputChars: 0, }; const hashlineSubtypes: Record = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0])); @@ -630,9 +733,16 @@ async function runSingleTask( const originalFiles = await collectOriginalFileContents(cwd, task.files); try { + const sessionSetup = await prepareBenchmarkSessionSetup({ + config, + task, + cwd, + expectedDir, + multiFile: false, + }); await fs.promises.appendFile( logFile, - `{"type":"meta","task":"${task.id}","run":${runIndex},"workDir":"${cwd}"}\n`, + `{"type":"meta","task":"${task.id}","run":${runIndex},"workDir":"${cwd}","providerSessionId":${JSON.stringify(sessionSetup.providerSessionId)}}\n`, ); const env: Record = { PI_NO_TITLE: "1" }; @@ -652,7 +762,7 @@ async function runSingleTask( cwd, provider: config.provider, model: config.model, - args: ["--tools", "read,edit,write", "--no-skills", "--no-title", "--no-rules"], + args: sessionSetup.rpcArgs, env, }); @@ -673,24 +783,25 @@ async function runSingleTask( let allEvents: Array<{ type: string; [key: string]: unknown }> = []; for (let attempt = 0; attempt < maxAttempts; attempt++) { - const guidedContext = await buildGuidedContext(task, cwd, expectedDir, config); - const promptWithContext = buildBenchmarkPrompt({ - multiFile: false, + const guidedContext = + attempt === 0 + ? sessionSetup.initialGuidedContext + : await buildGuidedContext(task, cwd, expectedDir, config); + const delivery = buildBenchmarkPromptDelivery({ taskPrompt: task.prompt, guidedContext, retryContext, - config, }); await fs.promises.appendFile( logFile, - `{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`, + `{"type":"prompt","attempt":${attempt + 1},"delivery":${JSON.stringify(delivery.kind)},"message":${JSON.stringify(delivery.message)}}\n`, ); const statsBefore = await client.getSessionStats(); let events: Array<{ type: string; [key: string]: unknown }>; try { - events = await collectPromptEvents(client, promptWithContext, config, logEvent); + events = await collectPromptEvents(client, delivery, config, logEvent); } catch (err) { if (err instanceof PromptTurnLimitError) { error = err.message; @@ -806,6 +917,15 @@ async function runSingleTask( editFailures.push({ toolCallId: e.toolCallId, args, error }); } else { toolStats.editSuccesses++; + const warningMessages = extractHashlineWarnings(e.result); + if (warningMessages.length > 0) { + editWarnings.push(...warningMessages); + toolStats.editWarnings += warningMessages.length; + if (hasHashlineAutocorrectWarning(warningMessages)) { + editAutocorrectCount++; + toolStats.editAutocorrects++; + } + } } } } @@ -892,6 +1012,8 @@ async function runSingleTask( diff, toolCalls: toolStats, editFailures, + editWarnings, + editAutocorrectCount, hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined, mutationIntentMatched: mutationIntentValidation?.matched, mutationIntentReason: mutationIntentValidation?.reason, @@ -920,6 +1042,8 @@ async function runBatchedTask( let agentResponse: string | undefined; let diff: string | undefined; const editFailures: EditFailure[] = []; + const editWarnings: string[] = []; + let editAutocorrectCount = 0; let timeoutTelemetry: PromptAttemptTelemetry | undefined; let mutationIntentValidation: MutationIntentValidation | null = null; const toolStats = { @@ -928,6 +1052,8 @@ async function runBatchedTask( write: 0, editSuccesses: 0, editFailures: 0, + editWarnings: 0, + editAutocorrects: 0, totalInputChars: 0, }; const hashlineSubtypes: Record = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0])); @@ -955,22 +1081,21 @@ async function runBatchedTask( for (let attempt = 0; attempt < maxAttempts; attempt++) { const guidedContext = await buildGuidedContext(task, cwd, expectedDir, config); - const promptWithContext = buildBenchmarkPrompt({ - multiFile: true, + const delivery = buildBenchmarkPromptDelivery({ taskPrompt: task.prompt, guidedContext, retryContext, - config, }); + await fs.promises.appendFile( logFile, - `{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`, + `{"type":"prompt","attempt":${attempt + 1},"delivery":${JSON.stringify(delivery.kind)},"message":${JSON.stringify(delivery.message)}}\n`, ); const statsBefore = await client.getSessionStats(); let events: Array<{ type: string; [key: string]: unknown }>; try { - events = await collectPromptEvents(client, promptWithContext, config, logEvent); + events = await collectPromptEvents(client, delivery, config, logEvent); } catch (err) { if (err instanceof PromptTurnLimitError) { error = err.message; @@ -1083,6 +1208,15 @@ async function runBatchedTask( editFailures.push({ toolCallId: e.toolCallId, args, error: toolError }); } else { toolStats.editSuccesses++; + const warningMessages = extractHashlineWarnings(e.result); + if (warningMessages.length > 0) { + editWarnings.push(...warningMessages); + toolStats.editWarnings += warningMessages.length; + if (hasHashlineAutocorrectWarning(warningMessages)) { + editAutocorrectCount++; + toolStats.editAutocorrects++; + } + } } } } @@ -1171,6 +1305,8 @@ async function runBatchedTask( diff, toolCalls: toolStats, editFailures, + editWarnings, + editAutocorrectCount, hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined, mutationIntentMatched: mutationIntentValidation?.matched, mutationIntentReason: mutationIntentValidation?.reason, @@ -1178,18 +1314,40 @@ async function runBatchedTask( }; } -function extractToolErrorMessage(result: unknown): string { +function extractToolText(result: unknown): string | null { if (typeof result === "string") return result; - if (!result || typeof result !== "object") return "Unknown error"; + if (!result || typeof result !== "object") return null; const content = (result as { content?: unknown }).content; - if (Array.isArray(content)) { - for (const entry of content) { - if (!entry || typeof entry !== "object") continue; - if (!("text" in entry)) continue; - const text = (entry as { text?: unknown }).text; - if (typeof text === "string") return text; - } + if (!Array.isArray(content)) return null; + for (const entry of content) { + if (!entry || typeof entry !== "object") continue; + if (!("text" in entry)) continue; + const text = (entry as { text?: unknown }).text; + if (typeof text === "string") return text; } + return null; +} + +function extractHashlineWarnings(result: unknown): string[] { + const text = extractToolText(result); + if (!text) return []; + const marker = "Warnings:\n"; + const markerIndex = text.indexOf(marker); + if (markerIndex === -1) return []; + return text + .slice(markerIndex + marker.length) + .split("\n") + .map(line => line.trim()) + .filter(Boolean); +} + +function hasHashlineAutocorrectWarning(warnings: string[]): boolean { + return warnings.some(warning => warning.startsWith("Auto-corrected ")); +} + +function extractToolErrorMessage(result: unknown): string { + const text = extractToolText(result); + if (text) return text; try { return JSON.stringify(result); } catch { @@ -1251,7 +1409,7 @@ function buildRunBatches(items: TaskRunItem[]): TaskRunItem[][] { async function collectPromptEvents( client: RpcClient, - prompt: string, + delivery: BenchmarkPromptDelivery, config: BenchmarkConfig, logEvent: (event: unknown) => Promise, ): Promise> { @@ -1367,7 +1525,11 @@ async function collectPromptEvents( }); try { - await client.prompt(prompt); + if (delivery.kind === "followUp") { + await client.followUp(delivery.message); + } else { + await client.prompt(delivery.message); + } } catch (err) { if (timer) { clearTimeout(timer); @@ -1419,13 +1581,26 @@ function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult { write: orderedRuns.reduce((sum, r) => sum + r.toolCalls.write, 0) / n, editSuccesses: orderedRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0) / n, editFailures: orderedRuns.reduce((sum, r) => sum + r.toolCalls.editFailures, 0) / n, + editWarnings: orderedRuns.reduce((sum, r) => sum + r.toolCalls.editWarnings, 0) / n, + editAutocorrects: orderedRuns.reduce((sum, r) => sum + r.toolCalls.editAutocorrects, 0) / n, totalInputChars: orderedRuns.reduce((sum, r) => sum + r.toolCalls.totalInputChars, 0) / n, } - : { read: 0, edit: 0, write: 0, editSuccesses: 0, editFailures: 0, totalInputChars: 0 }; + : { + read: 0, + edit: 0, + write: 0, + editSuccesses: 0, + editFailures: 0, + editWarnings: 0, + editAutocorrects: 0, + totalInputChars: 0, + }; const totalEditAttempts = orderedRuns.reduce((sum, r) => sum + r.toolCalls.edit, 0); const totalEditSuccesses = orderedRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0); const editSuccessRate = totalEditAttempts > 0 ? totalEditSuccesses / totalEditAttempts : 1; + const autocorrectFreeSuccesses = orderedRuns.filter(run => run.success && run.editAutocorrectCount === 0).length; + const autocorrectFreeSuccessRate = n > 0 ? autocorrectFreeSuccesses / n : 0; return { id: task.id, @@ -1438,6 +1613,7 @@ function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult { avgIndentScore, avgToolCalls, editSuccessRate, + autocorrectFreeSuccessRate, }; } @@ -1456,9 +1632,13 @@ function buildFailureResult(item: TaskRunItem, error: string): TaskRunResult { write: 0, editSuccesses: 0, editFailures: 0, + editWarnings: 0, + editAutocorrects: 0, totalInputChars: 0, }, editFailures: [], + editWarnings: [], + editAutocorrectCount: 0, }; } @@ -1488,13 +1668,21 @@ async function runBatch( env.PI_EDIT_FUZZY_THRESHOLD = config.editFuzzyThreshold === "auto" ? "auto" : String(config.editFuzzyThreshold); } + // Batches are intentionally size 1, so the provider cache key stays task-scoped. + const sessionSetup = await prepareBenchmarkSessionSetup({ + config, + task: orderedItems[0]!.task, + cwd: workDir, + expectedDir: orderedItems[0]!.task.expectedDir, + multiFile: true, + }); using client = new RpcClient({ cliPath: CLI_PATH, cwd: workDir, provider: config.provider, model: config.model, - args: ["--tools", "read,edit,write", "--no-skills", "--no-title", "--no-rules"], + args: sessionSetup.rpcArgs, sessionDir, env, }); @@ -1604,10 +1792,16 @@ export async function runBenchmark( write: allRuns.reduce((sum, r) => sum + r.toolCalls.write, 0), editSuccesses: allRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0), editFailures: allRuns.reduce((sum, r) => sum + r.toolCalls.editFailures, 0), + editWarnings: allRuns.reduce((sum, r) => sum + r.toolCalls.editWarnings, 0), + editAutocorrects: allRuns.reduce((sum, r) => sum + r.toolCalls.editAutocorrects, 0), totalInputChars: allRuns.reduce((sum, r) => sum + r.toolCalls.totalInputChars, 0), }; const editSuccessRate = totalToolCalls.edit > 0 ? totalToolCalls.editSuccesses / totalToolCalls.edit : 1; + const autocorrectFreeSuccessfulRuns = allRuns.filter(run => run.success && run.editAutocorrectCount === 0).length; + const autocorrectedRuns = allRuns.filter(run => run.editAutocorrectCount > 0).length; + const editAutocorrectRate = + totalToolCalls.editSuccesses > 0 ? totalToolCalls.editAutocorrects / totalToolCalls.editSuccesses : 0; const timeoutRuns = allRuns.filter(r => r.error?.includes("Timeout waiting for agent_end")).length; const runsWithMutationIntent = allRuns.filter(r => typeof r.mutationIntentMatched === "boolean"); const mutationIntentMatchRate = @@ -1648,9 +1842,15 @@ export async function runBenchmark( write: totalToolCalls.write / totalRuns, editSuccesses: totalToolCalls.editSuccesses / totalRuns, editFailures: totalToolCalls.editFailures / totalRuns, + editWarnings: totalToolCalls.editWarnings / totalRuns, + editAutocorrects: totalToolCalls.editAutocorrects / totalRuns, totalInputChars: totalToolCalls.totalInputChars / totalRuns, }, editSuccessRate, + autocorrectFreeSuccessfulRuns, + autocorrectFreeSuccessRate: autocorrectFreeSuccessfulRuns / totalRuns, + autocorrectedRuns, + editAutocorrectRate, timeoutRuns, mutationIntentMatchRate, hashlineEditSubtypes, From 66a478e6e6ebdb6805e9e5eb24381d982e63d698 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:14:17 +0100 Subject: [PATCH 12/22] feat(coding-agent): implemented dynamic tool activation for autoresearch lifecycle management - Added `defaultInactive` property to ToolDefinition for conditional tool registration and activation control. - Added dynamic tool activation/deactivation API `setActiveTools()` for managing experiment tools in autoresearch mode. - Replaced single `command-start.md` workflow with separate `command-initialize.md` and `command-resume.md` prompts for autoresearch initialization and session resumption. - Added interactive intent dialog for autoresearch optimization goals with automatic session resumption detection based on autoresearch.md presence. - Refactored autoresearch command handler to distinguish resume vs initialize flows and dynamically activate/deactivate experiment tools based on mode and session state. --- packages/coding-agent/CHANGELOG.md | 11 + .../src/autoresearch/command-initialize.md | 13 + .../src/autoresearch/command-resume.md | 9 + .../src/autoresearch/command-start.md | 10 - .../coding-agent/src/autoresearch/index.ts | 64 ++++- .../src/autoresearch/tools/init-experiment.ts | 1 + .../src/autoresearch/tools/log-experiment.ts | 1 + .../src/autoresearch/tools/run-experiment.ts | 1 + .../src/extensibility/extensions/types.ts | 3 + .../test/autoresearch-state.test.ts | 264 ++++++++++++++++++ .../test/sdk-tool-activation.test.ts | 108 +++++++ 11 files changed, 463 insertions(+), 22 deletions(-) create mode 100644 packages/coding-agent/src/autoresearch/command-initialize.md create mode 100644 packages/coding-agent/src/autoresearch/command-resume.md delete mode 100644 packages/coding-agent/src/autoresearch/command-start.md create mode 100644 packages/coding-agent/test/sdk-tool-activation.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 68513549f..1981256bd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Changed hashline edit operation types from `replace` (with optional `end`) to explicit `replace_line` and `replace_range` operations @@ -9,6 +10,11 @@ ### Added +- Added `defaultInactive` property to `ToolDefinition` to allow tools to be registered but excluded from the initial active set, with extension responsibility for activation/deactivation +- Added dynamic tool activation/deactivation in autoresearch mode via `setActiveTools()` API +- Added separate initialization and resume workflows for autoresearch with `command-initialize.md` and `command-resume.md` prompts +- Added intent dialog to prompt users for autoresearch optimization goals when starting fresh +- Added automatic detection of existing `autoresearch.md` to resume from previous sessions without re-prompting for intent - Added autoresearch extension with autonomous experiment loop capabilities - Added `init_experiment` tool to initialize and reset autoresearch sessions with configurable metrics - Added `log_experiment` tool to record experiment results with metric parsing and confidence tracking @@ -30,6 +36,10 @@ ### Changed +- Changed autoresearch command to use intent-based initialization instead of goal parameter, with user input dialog for new sessions +- Changed autoresearch startup to activate experiment tools (`init_experiment`, `run_experiment`, `log_experiment`) only when autoresearch mode is enabled +- Changed autoresearch shutdown to deactivate experiment tools when mode is disabled or cleared +- Changed autoresearch session rehydration to dynamically manage experiment tool activation based on session state - Refactored hashline edit validation to enforce stricter anchor requirements per operation type - Updated edit application logic to handle explicit file-level operations (`append_eof`, `prepend_bof`) separately from anchor-based operations - Changed `setWidget` API to accept `ExtensionWidgetOptions` parameter for placement control @@ -47,6 +57,7 @@ ### Removed +- Removed `command-start.md` prompt template in favor of separate initialize and resume workflows - Removed auto-correction of off-by-one range edits that duplicated closing braces or boundary lines - Removed `shouldAutocorrect` function and related boundary line deduplication logic from hashline editor - Removed auto-correction of off-by-one range edits that duplicated closing braces or boundary lines diff --git a/packages/coding-agent/src/autoresearch/command-initialize.md b/packages/coding-agent/src/autoresearch/command-initialize.md new file mode 100644 index 000000000..d4791b1d2 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/command-initialize.md @@ -0,0 +1,13 @@ +Set up autoresearch for this intent: + +{{intent}} + +Explain briefly what autoresearch will do in this repository, then initialize the workspace. + +Your first actions: +- write `autoresearch.md` +- define the benchmark entrypoint in `autoresearch.sh` +- optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate +- run `init_experiment` +- run and log the baseline +- keep iterating until interrupted or until the configured iteration cap is reached diff --git a/packages/coding-agent/src/autoresearch/command-resume.md b/packages/coding-agent/src/autoresearch/command-resume.md new file mode 100644 index 000000000..2ad853a94 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/command-resume.md @@ -0,0 +1,9 @@ +Resume autoresearch from the attached notes. + +@{{autoresearch_md_path}} + +Use the notes as the source of truth for the current direction. +- inspect recent git history for context +- inspect `autoresearch.jsonl` if it exists +- continue the most promising unfinished branch +- keep iterating until interrupted or until the configured iteration cap is reached diff --git a/packages/coding-agent/src/autoresearch/command-start.md b/packages/coding-agent/src/autoresearch/command-start.md deleted file mode 100644 index 1594a6a2c..000000000 --- a/packages/coding-agent/src/autoresearch/command-start.md +++ /dev/null @@ -1,10 +0,0 @@ -Autoresearch mode is active. - -Goal: -{{goal}} - -Start or resume the autoresearch loop now. - -- Read `autoresearch.md` if it already exists. -- Otherwise create the autoresearch workspace, initialize the experiment, run a baseline, and keep iterating. -- Continue until interrupted or until the configured iteration cap is reached. diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 126ec2701..8f13e5cdc 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -3,7 +3,8 @@ import * as path from "node:path"; import type { AutocompleteItem } from "@oh-my-pi/pi-tui"; import { renderPromptTemplate } from "../config/prompt-templates"; import type { ExtensionContext, ExtensionFactory } from "../extensibility/extensions"; -import commandStartTemplate from "./command-start.md" with { type: "text" }; +import commandInitializeTemplate from "./command-initialize.md" with { type: "text" }; +import commandResumeTemplate from "./command-resume.md" with { type: "text" }; import { createDashboardController } from "./dashboard"; import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "./helpers"; import promptTemplate from "./prompt.md" with { type: "text" }; @@ -22,6 +23,7 @@ import type { AutoresearchRuntime } from "./types"; const AUTORESUME_INTERVAL_MS = 5 * 60 * 1000; const MAX_AUTORESUME_TURNS = 20; +const EXPERIMENT_TOOL_NAMES = ["init_experiment", "run_experiment", "log_experiment"]; export const createAutoresearchExtension: ExtensionFactory = api => { const runtimeStore = createRuntimeStore(); @@ -30,7 +32,7 @@ export const createAutoresearchExtension: ExtensionFactory = api => { const getSessionKey = (ctx: ExtensionContext): string => ctx.sessionManager.getSessionId(); const getRuntime = (ctx: ExtensionContext): AutoresearchRuntime => runtimeStore.ensure(getSessionKey(ctx)); - const rehydrate = (ctx: ExtensionContext): void => { + const rehydrate = async (ctx: ExtensionContext): Promise => { const runtime = getRuntime(ctx); const workDir = resolveWorkDir(ctx.cwd); const reconstructed = reconstructStateFromJsonl(workDir); @@ -47,6 +49,17 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.lastRunAsi = null; runtime.runningExperiment = null; dashboard.updateWidget(ctx, runtime); + const activeTools = api.getActiveTools(); + const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); + const nextActiveTools = runtime.autoresearchMode + ? [...new Set([...activeTools, ...EXPERIMENT_TOOL_NAMES])] + : activeTools.filter(name => !experimentTools.has(name)); + const toolsChanged = + nextActiveTools.length !== activeTools.length || + nextActiveTools.some((name, index) => name !== activeTools[index]); + if (toolsChanged) { + await api.setActiveTools(nextActiveTools); + } }; const setMode = ( @@ -86,15 +99,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => { return; } - if (trimmed.length === 0) { - ctx.ui.notify("Usage: /autoresearch | off | clear", "info"); - return; - } if (trimmed === "off") { setMode(ctx, false, runtime.goal, "off"); runtime.experimentsThisSession = 0; runtime.autoResumeTurns = 0; dashboard.updateWidget(ctx, runtime); + const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); + await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); ctx.ui.notify("Autoresearch mode disabled", "info"); return; } @@ -109,19 +120,48 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.goal = null; setMode(ctx, false, null, "clear"); dashboard.updateWidget(ctx, runtime); + const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); + await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); ctx.ui.notify("Autoresearch log cleared", "info"); return; } - setMode(ctx, true, trimmed, "on"); + const workDir = resolveWorkDir(ctx.cwd); + const autoresearchMdPath = path.join(workDir, "autoresearch.md"); + const hasAutoresearchMd = fs.existsSync(autoresearchMdPath); + + if (hasAutoresearchMd) { + setMode(ctx, true, runtime.goal, "on"); + runtime.experimentsThisSession = 0; + runtime.autoResumeTurns = 0; + dashboard.updateWidget(ctx, runtime); + await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); + api.sendUserMessage( + renderPromptTemplate(commandResumeTemplate, { + autoresearch_md_path: autoresearchMdPath, + }), + ); + return; + } + + const intentInput = await ctx.ui.input( + "Autoresearch Intent", + trimmed || runtime.goal || "what should autoresearch improve?", + ); + if (intentInput === undefined) return; + + const intent = intentInput.trim(); + if (intent.length === 0) { + ctx.ui.notify("Autoresearch intent is required", "info"); + return; + } + + setMode(ctx, true, intent, "on"); runtime.experimentsThisSession = 0; runtime.autoResumeTurns = 0; dashboard.updateWidget(ctx, runtime); - api.sendUserMessage( - renderPromptTemplate(commandStartTemplate, { - goal: trimmed, - }), - ); + await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); + api.sendUserMessage(renderPromptTemplate(commandInitializeTemplate, { intent })); }, }); diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts index b6aa8a659..19744fd2d 100644 --- a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -41,6 +41,7 @@ export function createInitExperimentTool( description: "Initialize or reset the autoresearch session for the current optimization target before the first logged run of a segment.", parameters: initExperimentSchema, + defaultInactive: true, async execute(_toolCallId, params, _signal, _onUpdate, ctx) { const workDirError = validateWorkDir(ctx.cwd); if (workDirError) { diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index 8fb002b5a..e0dfa0e53 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -74,6 +74,7 @@ export function createLogExperimentTool( description: "Log the experiment result, update dashboard state, persist JSONL history, and apply git keep or revert behavior.", parameters: logExperimentSchema, + defaultInactive: true, async execute(_toolCallId, params, _signal, _onUpdate, ctx) { const workDirError = validateWorkDir(ctx.cwd); if (workDirError) { diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts index 993a6b2d3..a5e502b15 100644 --- a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -61,6 +61,7 @@ export function createRunExperimentTool( description: "Run an experiment command with timing, tail capture, structured metric parsing, and optional autoresearch.checks.sh validation.", parameters: runExperimentSchema, + defaultInactive: true, async execute(_toolCallId, params, signal, onUpdate, ctx) { const workDirError = validateWorkDir(ctx.cwd); if (workDirError) { diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index f670e147c..6f556a6dd 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -307,6 +307,9 @@ export interface ToolDefinition { expect(isAutoresearchShCommand("bash -lc 'autoresearch.sh'")).toBe(false); }); }); + +interface AutoresearchCommandHarness { + command: RegisteredCommand; + ctx: ExtensionCommandContext; + sentMessages: string[]; + inputCalls: Array<{ title: string; placeholder: string | undefined }>; + notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }>; +} + +function createAutoresearchCommandHarness(cwd: string, inputResult: string | undefined): AutoresearchCommandHarness { + const sentMessages: string[] = []; + const inputCalls: Array<{ title: string; placeholder: string | undefined }> = []; + const notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }> = []; + let command: RegisteredCommand | undefined; + + const api = { + appendEntry(_customType: string, _data?: unknown): void {}, + on(): void {}, + registerCommand(name: string, options: Omit): void { + command = { name, ...options }; + }, + registerShortcut(): void {}, + registerTool(): void {}, + getActiveTools(): string[] { + return []; + }, + setActiveTools: async (_toolNames: string[]): Promise => {}, + sendUserMessage(content: string | unknown[]): void { + if (typeof content !== "string") { + throw new Error("Expected autoresearch command to send plain text"); + } + sentMessages.push(content); + }, + } as unknown as ExtensionAPI; + createAutoresearchExtension(api); + if (!command) throw new Error("Expected autoresearch command to register"); + + const ctx = { + abort(): void {}, + branch: async () => ({ cancelled: false }), + compact: async () => {}, + cwd, + getContextUsage: () => undefined, + hasUI: false, + isIdle: () => true, + model: undefined, + modelRegistry: {}, + newSession: async () => ({ cancelled: false }), + reload: async () => {}, + sessionManager: { + getEntries: () => [], + getSessionId: () => "session-1", + }, + switchSession: async () => ({ cancelled: false }), + navigateTree: async () => ({ cancelled: false }), + ui: { + confirm: async () => false, + custom: async () => undefined, + input: async (title: string, placeholder?: string) => { + inputCalls.push({ title, placeholder }); + return inputResult; + }, + notify(message: string, type?: "info" | "warning" | "error"): void { + notifications.push({ message, type }); + }, + onTerminalInput: () => () => {}, + select: async () => undefined, + setFooter(): void {}, + setHeader(): void {}, + setStatus(): void {}, + setTitle(): void {}, + setWidget(): void {}, + setWorkingMessage(): void {}, + }, + waitForIdle: async () => {}, + } as unknown as ExtensionCommandContext; + + return { command, ctx, sentMessages, inputCalls, notifications }; +} + +interface AutoresearchLifecycleHarness { + sessionStartHandler: ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) | undefined; + sessionSwitchHandler: ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined; + ctx: ExtensionContext; + setActiveToolsCalls: string[][]; +} + +function createAutoresearchLifecycleHarness(options: { + activeTools: string[]; + controlEntries?: Array<{ type: "custom"; customType: string; data?: unknown }>; +}): AutoresearchLifecycleHarness { + const handlers = new Map Promise | void>(); + const activeTools = [...options.activeTools]; + const setActiveToolsCalls: string[][] = []; + + const api = { + appendEntry(_customType: string, _data?: unknown): void {}, + on(event: string, handler: (...args: unknown[]) => Promise | void): void { + handlers.set(event, handler); + }, + registerCommand(): void {}, + registerShortcut(): void {}, + registerTool(): void {}, + getActiveTools(): string[] { + return [...activeTools]; + }, + async setActiveTools(toolNames: string[]): Promise { + setActiveToolsCalls.push([...toolNames]); + activeTools.splice(0, activeTools.length, ...toolNames); + }, + sendUserMessage(): void {}, + } as unknown as ExtensionAPI; + createAutoresearchExtension(api); + + const ctx = { + abort(): void {}, + compact: async () => {}, + cwd: makeTempDir(), + getContextUsage: () => undefined, + hasUI: false, + hasPendingMessages: () => false, + isIdle: () => true, + model: undefined, + modelRegistry: {}, + sessionManager: { + getEntries: () => options.controlEntries ?? [], + getSessionId: () => "session-1", + }, + shutdown: async () => {}, + ui: { + confirm: async () => false, + custom: async () => undefined, + editor: async () => undefined, + getEditorText: () => "", + input: async () => undefined, + notify(): void {}, + onTerminalInput: () => () => {}, + select: async () => undefined, + setEditorComponent(): void {}, + setEditorText(): void {}, + setFooter(): void {}, + setHeader(): void {}, + setStatus(): void {}, + setTheme: async () => false, + setTitle(): void {}, + setToolsExpanded(): void {}, + setWidget(): void {}, + setWorkingMessage(): void {}, + }, + } as unknown as ExtensionContext; + + return { + sessionStartHandler: handlers.get("session_start") as + | ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) + | undefined, + sessionSwitchHandler: handlers.get("session_switch") as + | ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) + | undefined, + ctx, + setActiveToolsCalls, + }; +} + +describe("autoresearch command startup", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("asks for intent and sends an initialization prompt when no autoresearch.md exists", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const harness = createAutoresearchCommandHarness(dir, "reduce edit benchmark runtime variance"); + + await harness.command.handler("", harness.ctx); + + expect(harness.inputCalls).toEqual([ + { title: "Autoresearch Intent", placeholder: "what should autoresearch improve?" }, + ]); + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); + expect(harness.sentMessages[0]).toContain("reduce edit benchmark runtime variance"); + expect(harness.sentMessages[0]).toContain("Explain briefly what autoresearch will do in this repository"); + expect(harness.notifications).toEqual([]); + }); + + it("resumes from autoresearch.md without asking for intent when notes already exist", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const autoresearchMdPath = path.join(dir, "autoresearch.md"); + fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); + const harness = createAutoresearchCommandHarness(dir, "ignored"); + + await harness.command.handler("", harness.ctx); + + expect(harness.inputCalls).toEqual([]); + expect(harness.sentMessages).toEqual([ + [ + "Resume autoresearch from the attached notes.", + "", + `@${autoresearchMdPath}`, + "", + "Use the notes as the source of truth for the current direction.", + "- inspect recent git history for context", + "- inspect `autoresearch.jsonl` if it exists", + "- continue the most promising unfinished branch", + "- keep iterating until interrupted or until the configured iteration cap is reached", + ].join("\n"), + ]); + }); + + it("does not start autoresearch when the intent dialog returns blank input", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const harness = createAutoresearchCommandHarness(dir, " "); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([{ message: "Autoresearch intent is required", type: "info" }]); + }); +}); + +describe("autoresearch lifecycle tool activation", () => { + it("activates experiment tools when rehydrating an autoresearch session", async () => { + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["read", "write"], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "speed" } }], + }); + + if (!harness.sessionStartHandler) throw new Error("Expected session_start handler"); + await harness.sessionStartHandler({ type: "session_start" }, harness.ctx); + + expect(harness.setActiveToolsCalls).toEqual([ + ["read", "write", "init_experiment", "run_experiment", "log_experiment"], + ]); + }); + + it("removes experiment tools when rehydrating a non-autoresearch session", async () => { + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["read", "init_experiment", "run_experiment", "log_experiment"], + }); + + if (!harness.sessionSwitchHandler) throw new Error("Expected session_switch handler"); + await harness.sessionSwitchHandler( + { type: "session_switch", reason: "resume", previousSessionFile: "/tmp/previous.jsonl" }, + harness.ctx, + ); + + expect(harness.setActiveToolsCalls).toEqual([["read"]]); + }); +}); diff --git a/packages/coding-agent/test/sdk-tool-activation.test.ts b/packages/coding-agent/test/sdk-tool-activation.test.ts new file mode 100644 index 000000000..cc8febb87 --- /dev/null +++ b/packages/coding-agent/test/sdk-tool-activation.test.ts @@ -0,0 +1,108 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { getBundledModel } from "@oh-my-pi/pi-ai"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { createAgentSession, type ExtensionFactory } from "@oh-my-pi/pi-coding-agent/sdk"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { Type } from "@sinclair/typebox"; + +const toolActivationExtension: ExtensionFactory = pi => { + pi.registerTool({ + name: "default_inactive_tool", + label: "Default Inactive Tool", + description: "Tool hidden from the initial active set unless explicitly requested.", + parameters: Type.Object({}), + defaultInactive: true, + async execute() { + return { content: [{ type: "text", text: "inactive" }] }; + }, + }); + pi.registerTool({ + name: "default_active_tool", + label: "Default Active Tool", + description: "Tool included in the initial active set.", + parameters: Type.Object({}), + async execute() { + return { content: [{ type: "text", text: "active" }] }; + }, + }); +}; + +describe("createAgentSession defaultInactive tool activation", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const tempDir of tempDirs.splice(0)) { + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }); + + it("excludes defaultInactive extension tools from the initial active set unless explicitly requested", async () => { + const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); + tempDirs.push(tempDir); + fs.mkdirSync(tempDir, { recursive: true }); + + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated(), + model: getBundledModel("openai", "gpt-4o-mini"), + disableExtensionDiscovery: true, + extensions: [toolActivationExtension], + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + }); + + try { + expect(session.getAllToolNames()).toEqual( + expect.arrayContaining(["default_active_tool", "default_inactive_tool"]), + ); + expect(session.getActiveToolNames()).toContain("default_active_tool"); + expect(session.getActiveToolNames()).not.toContain("default_inactive_tool"); + expect(session.systemPrompt).toContain("default_active_tool"); + expect(session.systemPrompt).not.toContain("default_inactive_tool"); + } finally { + await session.dispose(); + } + }); + + it("allows explicitly requested defaultInactive extension tools into the initial active set", async () => { + const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); + tempDirs.push(tempDir); + fs.mkdirSync(tempDir, { recursive: true }); + + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated(), + model: getBundledModel("openai", "gpt-4o-mini"), + disableExtensionDiscovery: true, + extensions: [toolActivationExtension], + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + toolNames: ["read", "default_inactive_tool"], + }); + + try { + expect(session.getActiveToolNames()).toEqual( + expect.arrayContaining(["read", "default_active_tool", "default_inactive_tool"]), + ); + expect(session.systemPrompt).toContain("default_inactive_tool"); + } finally { + await session.dispose(); + } + }); +}); From 012e0c90b9f947ef004e88b86fb79051c99d2b8b Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:14:31 +0100 Subject: [PATCH 13/22] feat(coding-agent): renamed hashline operation types for clarity - Renamed hashline operation types for clarity: append->append_at, prepend->prepend_at, append_eof->append_file, prepend_bof->prepend_file. - Updated all operation type references in patch implementation, tests, and documentation to reflect new naming convention. - Restructured hashline tool documentation with hierarchical sections and simplified examples for improved clarity. - Consolidated validation rules and added explicit warning about invalid anchors and operation/field combinations. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/patch/hashline.ts | 40 ++--- packages/coding-agent/src/patch/index.ts | 44 +++--- .../src/prompts/tools/hashline.md | 148 ++++++------------ .../coding-agent/test/core/hashline.test.ts | 36 ++--- 5 files changed, 106 insertions(+), 164 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1981256bd..bad9eab69 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,9 @@ # Changelog ## [Unreleased] - ### Breaking Changes +- Renamed hashline edit operation types: `append` → `append_at`, `prepend` → `prepend_at`, `append_eof` → `append_file`, `prepend_bof` → `prepend_file` - Changed hashline edit operation types from `replace` (with optional `end`) to explicit `replace_line` and `replace_range` operations - Added required `append_eof` and `prepend_bof` operations for file-level edits; `append` and `prepend` now require an anchor position - Made `pos` parameter required for `replace_line`, `append`, and `prepend` operations; `append_eof` and `prepend_bof` no longer accept anchors diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 89e1df7e7..e4c2b6df5 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -18,10 +18,10 @@ export type Anchor = { line: number; hash: string }; export type HashlineEdit = | { op: "replace_line"; pos: Anchor; lines: string[] } | { op: "replace_range"; pos: Anchor; end: Anchor; lines: string[] } - | { op: "append"; pos: Anchor; lines: string[] } - | { op: "prepend"; pos: Anchor; lines: string[] } - | { op: "append_eof"; lines: string[] } - | { op: "prepend_bof"; lines: string[] }; + | { op: "append_at"; pos: Anchor; lines: string[] } + | { op: "prepend_at"; pos: Anchor; lines: string[] } + | { op: "append_file"; lines: string[] } + | { op: "prepend_file"; lines: string[] }; const NIBBLE_STR = "ZPMQVRWSNKTXJBYH"; @@ -517,16 +517,16 @@ export function applyHashlineEdits( } break; } - case "append": - case "prepend": { + case "append_at": + case "prepend_at": { if (!validateRef(edit.pos)) continue; if (edit.lines.length === 0) { edit.lines = [""]; // insert an empty line } break; } - case "append_eof": - case "prepend_bof": { + case "append_file": + case "prepend_file": { if (edit.lines.length === 0) { edit.lines = [""]; // insert an empty line } @@ -552,16 +552,16 @@ export function applyHashlineEdits( case "replace_range": lineKey = `r:${edit.pos.line}:${edit.end.line}`; break; - case "append": + case "append_at": lineKey = `i:${edit.pos.line}`; break; - case "prepend": + case "prepend_at": lineKey = `ib:${edit.pos.line}`; break; - case "append_eof": + case "append_file": lineKey = "ieof"; break; - case "prepend_bof": + case "prepend_file": lineKey = "ibef"; break; } @@ -591,19 +591,19 @@ export function applyHashlineEdits( sortLine = edit.end.line; precedence = 0; break; - case "append": + case "append_at": sortLine = edit.pos.line; precedence = 1; break; - case "prepend": + case "prepend_at": sortLine = edit.pos.line; precedence = 2; break; - case "append_eof": + case "append_file": sortLine = fileLines.length + 1; precedence = 1; break; - case "prepend_bof": + case "prepend_file": sortLine = 0; precedence = 2; break; @@ -637,7 +637,7 @@ export function applyHashlineEdits( trackFirstChanged(edit.pos.line); break; } - case "append": { + case "append_at": { const inserted = edit.lines; if (inserted.length === 0) { noopEdits.push({ @@ -651,7 +651,7 @@ export function applyHashlineEdits( trackFirstChanged(edit.pos.line + 1); break; } - case "prepend": { + case "prepend_at": { const inserted = edit.lines; if (inserted.length === 0) { noopEdits.push({ @@ -665,7 +665,7 @@ export function applyHashlineEdits( trackFirstChanged(edit.pos.line); break; } - case "append_eof": { + case "append_file": { const inserted = edit.lines; if (inserted.length === 0) { noopEdits.push({ editIndex: idx, loc: "EOF", current: "" }); @@ -680,7 +680,7 @@ export function applyHashlineEdits( } break; } - case "prepend_bof": { + case "prepend_file": { const inserted = edit.lines; if (inserted.length === 0) { noopEdits.push({ editIndex: idx, loc: "BOF", current: "" }); diff --git a/packages/coding-agent/src/patch/index.ts b/packages/coding-agent/src/patch/index.ts index 38651e2a8..79ee06338 100644 --- a/packages/coding-agent/src/patch/index.ts +++ b/packages/coding-agent/src/patch/index.ts @@ -176,7 +176,7 @@ export function hashlineParseText(edit: string[] | string | null): string[] { const hashlineEditSchema = Type.Object( { - op: StringEnum(["replace_line", "replace_range", "append", "prepend", "append_eof", "prepend_bof"]), + op: StringEnum(["replace_line", "replace_range", "append_at", "prepend_at", "append_file", "prepend_file"]), pos: Type.Optional(Type.String({ description: "anchor" })), end: Type.Optional(Type.String({ description: "limit position" })), lines: Type.Union([ @@ -211,10 +211,10 @@ export type HashlineParams = Static; * Resilient: as long as at least one anchor exists, we execute. * - replace_line + tag → single-line replace * - replace_range + tag + end → range replace - * - append + tag or end → append after that anchor - * - prepend + tag or end → prepend before that anchor - * - append_eof → file-level append (no anchors needed) - * - prepend_bof → file-level prepend (no anchors needed) + * - append_at + tag or end → append after that anchor + * - prepend_at + tag or end → prepend before that anchor + * - append_file → file-level append (no anchors needed) + * - prepend_file → file-level prepend (no anchors needed) * * Unknown ops default to replace_line/replace_range based on available anchors. */ @@ -237,24 +237,24 @@ function resolveEditAnchors(edits: HashlineToolEdit[]): HashlineEdit[] { result.push({ op: "replace_range", pos: tag, end, lines }); break; } - case "append": { + case "append_at": { const anchor = tag ?? end; - if (!anchor) throw new Error("append requires an anchor (pos)."); - result.push({ op: "append", pos: anchor, lines }); + if (!anchor) throw new Error("append_at requires an anchor (pos)."); + result.push({ op: "append_at", pos: anchor, lines }); break; } - case "prepend": { + case "prepend_at": { const anchor = end ?? tag; - if (!anchor) throw new Error("prepend requires an anchor (pos)."); - result.push({ op: "prepend", pos: anchor, lines }); + if (!anchor) throw new Error("prepend_at requires an anchor (pos)."); + result.push({ op: "prepend_at", pos: anchor, lines }); break; } - case "append_eof": { - result.push({ op: "append_eof", lines }); + case "append_file": { + result.push({ op: "append_file", lines }); break; } - case "prepend_bof": { - result.push({ op: "prepend_bof", lines }); + case "prepend_file": { + result.push({ op: "prepend_file", lines }); break; } default: { @@ -575,8 +575,8 @@ export class EditTool implements AgentTool { const lines: string[] = []; for (const edit of edits) { // For file creation, only anchorless appends/prepends are valid - if (edit.op === "append_eof" || edit.op === "prepend_bof") { - if (edit.op === "prepend_bof") { + if (edit.op === "append_file" || edit.op === "prepend_file") { + if (edit.op === "prepend_file") { lines.unshift(...hashlineParseText(edit.lines)); } else { lines.push(...hashlineParseText(edit.lines)); @@ -605,7 +605,7 @@ export class EditTool implements AgentTool { const originalNormalized = normalizeToLF(text); let normalizedText = originalNormalized; - // Apply anchor-based edits first (replace, append, prepend) + // Apply anchor-based edits first (replace, append_at, prepend_at) const anchorResult = applyHashlineEdits(normalizedText, anchorEdits); normalizedText = anchorResult.lines; @@ -641,12 +641,12 @@ export class EditTool implements AgentTool { case "replace_range": refs.push(edit.end, edit.pos); break; - case "append": - case "prepend": + case "append_at": + case "prepend_at": refs.push(edit.pos); break; - case "append_eof": - case "prepend_bof": + case "append_file": + case "prepend_file": break; } diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index ce3342160..fcbc3f233 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -1,27 +1,36 @@ -Applies precise, surgical file edits by referencing `LINE#ID` tags from `read` output. Each tag uniquely identifies a line, so edits remain stable even when lines shift. +Applies precise file edits using `LINE#ID` anchors from `read` output. -Read the file first to get fresh tags. Submit one `edit` call per file with all operations batched — tags shift after each edit, so multiple calls require re-reading between them. +Read the file first. Copy anchors exactly from the latest `read` output. In one `edit` call, batch all edits for one file. After any successful edit, re-read before editing that file again. + +This matters: your output is checked against the real file state. Invalid anchors, invalid op/field combinations, duplicated boundary lines, or semantically equivalent rewrites will fail. -**`path`** — the path to the file to edit. -**`move`** — if set, move the file to the given path. -**`delete`** — if true, delete the file. +**Top level** +- `path` — file path +- `move` — optional rename target +- `delete` — optional whole-file delete +- `edits` — array of edit entries -**`edits[n].pos`** — the anchor line. Meaning depends on `op`: - - if `replace_line`: the line to rewrite - - if `replace_range`: first line of the range to rewrite - - if `prepend`: line to insert new lines **before** - - if `append`: line to insert new lines **after** - - Not used by `append_eof` or `prepend_bof`. -**`edits[n].end`** — only used by `replace_range`: the last line of the range (inclusive). -**`edits[n].lines`** — the replacement content: - - for `replace_line`/`replace_range`: the exact lines that will replace the target line(s) - - for `append`/`prepend`/`append_eof`/`prepend_bof`: the new lines to insert - - `[""]` — blank line - - `null` or `[]` — delete if `replace_line`/`replace_range` -- If `lines` contains content that already exists after `end`, those lines **will be duplicated** in the output. -- Keep `lines` to exactly what belongs inside the consumed range. -- Ops are applied bottom-up. Tags **MUST** be referenced from the most recent `read` output. +**Edit entry shape** +Each entry is: +- `op` — one of `replace_line`, `replace_range`, `append_at`, `prepend_at`, `append_file`, `prepend_file` +- `lines` — replacement/inserted content +- `pos` — required for `replace_line`, `replace_range`, `append_at`, `prepend_at` +- `end` — required only for `replace_range` + +**Meaning** +- `replace_line`: replace exactly one anchored line +- `replace_range`: replace inclusive `pos..end` +- `append_at`: insert after `pos` +- `prepend_at`: insert before `pos` +- `append_file`: insert at end of file +- `prepend_file`: insert at beginning of file + +**`lines`** +- Array of literal file lines is preferred +- `""` means a blank line +- `null` or `[]` deletes for `replace_line` / `replace_range` +- For insert ops, `lines` must contain only the new content @@ -47,8 +56,7 @@ All examples below reference the same file, `util.ts`: {{hlinefull 18 "}"}} ``` - -Change the timeout from `5000` to `30_000`: + ``` { path: "util.ts", @@ -61,19 +69,7 @@ Change the timeout from `5000` to `30_000`: ``` - -Single line — `lines: null` deletes entirely: -``` -{ - path: "util.ts", - edits: [{ - op: "replace_line", - pos: {{hlineref 1 "// @ts-ignore"}}, - lines: null - }] -} -``` -Range — remove the legacy block (lines 10–11): + ``` { path: "util.ts", @@ -87,10 +83,8 @@ Range — remove the legacy block (lines 10–11): ``` - -Replace the catch body with smarter error handling. Shape (a): `pos` is the first body line, `end` is the last body line. The catch header (line 14) and its closer (line 17) are outside the range and stay untouched. - -When changing body content, replace the **entire** body span — not just one line inside it. Patching one line leaves the rest of the body stale. + +Replace only the catch body. Do not target the shared boundary line `} catch (err) {`. ``` { path: "util.ts", @@ -107,60 +101,13 @@ When changing body content, replace the **entire** body span — not just one li ``` - -Simplify `beta()` to a one-liner. Shape (b): `pos`=header, `end`=closer, re-emit all in `lines`. - -Bad — `end` stops at the inner `\t}` on line 17, so the outer `}` on line 18 survives. Result: two consecutive `}` lines. + +When adding a sibling declaration, prefer `prepend_at` on the next declaration. ``` { path: "util.ts", edits: [{ - op: "replace_range", - pos: {{hlineref 9 "function beta() {"}}, - end: {{hlineref 17 "\t}"}}, - lines: [ - "function beta() {", - "\treturn parse(data);", - "}" - ] - }] -} -``` -Good — `end` includes the function's own `}` on line 18, so the old closer is consumed: -``` -{ - path: "util.ts", - edits: [{ - op: "replace_range", - pos: {{hlineref 9 "function beta() {"}}, - end: {{hlineref 18 "}"}}, - lines: [ - "function beta() {", - "\treturn parse(data);", - "}" - ] - }] -} -``` - - - -Do not anchor `replace_range` on a mixed boundary line such as `} catch (err) {`, `} else {`, `}),`, or `},{`. Those lines belong to two adjacent structures at once. - -Bad — if you need to change code on both sides of that line, replacing just the boundary span will usually leave one side's syntax behind. - -Good — choose one of two safe shapes instead: -- move inward and replace only body-owned lines -- expand outward and replace one whole owned block, consuming its real closer/separator too - - - -Add a `gamma()` function between `alpha()` and `beta()`. Use `prepend` on the next declaration — not `append` on the previous block's closing brace — so the anchor is a stable declaration boundary. -``` -{ - path: "util.ts", - edits: [{ - op: "prepend", + op: "prepend_at", pos: {{hlineref 9 "function beta() {"}}, lines: [ "function gamma() {", @@ -171,22 +118,17 @@ Add a `gamma()` function between `alpha()` and `beta()`. Use `prepend` on the ne }] } ``` -Use a trailing `""` to preserve the blank line between sibling declarations. -- You **MUST NOT** use this tool to reformat, reindent, or adjust whitespace — run the project's formatter instead. -- Every tag **MUST** be copied exactly from your most recent `read` output as `N#ID`. Stale or mistyped tags cause mismatches. -- Edit payload: `{ path, edits[] }`. Each entry: `op`, `lines`, plus `pos` and/or `end` depending on op. `replace_line`/`append`/`prepend` require `pos`. `replace_range` requires both `pos` and `end`. `append_eof`/`prepend_bof` require neither. No extra keys. -- For `append`/`prepend`/`append_eof`/`prepend_bof`, `lines` **MUST** contain only the newly introduced content. Do not re-emit surrounding content, or terminators that already exist. -- When changing existing code near a block tail or closing delimiter, default to `replace_range` over the owned span instead of inserting around the boundary. -- When adding a sibling declaration, default to `prepend` on the next sibling declaration instead of `append` on the previous block's closing brace. -- **Block boundaries travel together.** For a block `{ header / body / closer }`, there are exactly two valid replace shapes: (a) replace only the body — `pos`=first body line, `end`=last body line, leave the header and closer untouched; or (b) replace the whole block — `pos`=header, `end`=closer, re-emit all three in `lines`. Never split them: do not set `end` to the closer while omitting it from `lines` (deletes it), and do not emit the closer in `lines` without including it in `end` (duplicates it). This applies to every block terminator: `}`, `continue`, `break`, `return`, `throw`. -- **Never target shared boundary lines.** Do not use `replace_range` spans that start, end, or pivot on a line that closes one construct and opens/separates another, such as `},{`, `}),`, `} else {`, or `} catch (err) {`. Those lines are not owned by a single block. Move the range inward to body-only lines, or widen it to consume one whole owned construct including its true trailing delimiter. -- **`lines` must not extend past `end`.** `lines` replaces exactly `pos..end`. Content after `end` survives. If you include lines in `lines` that exist after `end`, they will appear twice. Either extend `end` to cover all lines you are re-emitting, or remove the extra lines from `lines`. -- `lines` entries **MUST** be literal file content with indentation copied exactly from the `read` output. If the file uses tabs, use a real tab character. -- After any successful `edit` call on a file, the next change to that same file **MUST** start with a fresh `read`. Do not chain a second `edit` call off stale mental state, even if the intended range is nearby. -- If you need a second change in the same local region, default to one wider `replace` over the whole owned block instead of a sequence of micro-edits on adjacent lines. Repeated small patches in a moving region are unstable. -- If a local region is already malformed or a prior patch partially landed, stop nibbling at it. Re-read the file and replace the full owned block from a stable boundary; for a small file, prefer rewriting the file over stacking more tiny repairs. +- Make the minimum exact edit. Do not rewrite nearby code unless the consumed range requires it. +- Use anchors exactly as `N#ID` from the latest `read` output. +- `replace_range` requires both `pos` and `end`. All other anchored ops require `pos` only. +- `append_file` and `prepend_file` do not take anchors. +- Replace exactly the owned span. If `lines` re-emits content beyond `end`, it will duplicate. +- Do not target shared boundary lines such as `} else {`, `} catch (…) {`, `}),`, or `},{`. +- For a block, either replace only the body or replace the whole block. Do not split block boundaries. +- `lines` must be literal file content with matching indentation. If the file uses tabs, use real tabs. +- Do not use this tool to reformat or clean up unrelated code. \ No newline at end of file diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 1e6e61187..d1edbccd3 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -350,7 +350,7 @@ describe("applyHashlineEdits — delete", () => { describe("applyHashlineEdits — append", () => { it("inserts after a line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "append", pos: makeTag(1, "aaa"), lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "append_at", pos: makeTag(1, "aaa"), lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nNEW\nbbb\nccc"); @@ -359,7 +359,7 @@ describe("applyHashlineEdits — append", () => { it("inserts multiple lines", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "append", pos: makeTag(1, "aaa"), lines: ["x", "y", "z"] }]; + const edits: HashlineEdit[] = [{ op: "append_at", pos: makeTag(1, "aaa"), lines: ["x", "y", "z"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nx\ny\nz\nbbb"); @@ -367,7 +367,7 @@ describe("applyHashlineEdits — append", () => { it("inserts after last line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "append", pos: makeTag(2, "bbb"), lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "append_at", pos: makeTag(2, "bbb"), lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb\nNEW"); @@ -375,7 +375,7 @@ describe("applyHashlineEdits — append", () => { it("insert with empty dst inserts an empty line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "append", pos: makeTag(1, "aaa"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "append_at", pos: makeTag(1, "aaa"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\n\nbbb"); @@ -384,7 +384,7 @@ describe("applyHashlineEdits — append", () => { it("inserts at EOF without anchors", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "append_eof", lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "append_file", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb\nNEW"); @@ -393,7 +393,7 @@ describe("applyHashlineEdits — append", () => { it("inserts at EOF into empty file without anchors", () => { const content = ""; - const edits: HashlineEdit[] = [{ op: "append_eof", lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "append_file", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("NEW"); @@ -402,7 +402,7 @@ describe("applyHashlineEdits — append", () => { it("insert at EOF with empty dst inserts a trailing empty line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "append_eof", lines: [] }]; + const edits: HashlineEdit[] = [{ op: "append_file", lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nbbb\n"); @@ -417,7 +417,7 @@ describe("applyHashlineEdits — append", () => { describe("applyHashlineEdits — prepend", () => { it("inserts before a line", () => { const content = "aaa\nbbb\nccc"; - const edits: HashlineEdit[] = [{ op: "prepend", pos: makeTag(2, "bbb"), lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "prepend_at", pos: makeTag(2, "bbb"), lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nNEW\nbbb\nccc"); expect(result.firstChangedLine).toBe(2); @@ -425,21 +425,21 @@ describe("applyHashlineEdits — prepend", () => { it("inserts multiple lines before", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "prepend", pos: makeTag(2, "bbb"), lines: ["x", "y", "z"] }]; + const edits: HashlineEdit[] = [{ op: "prepend_at", pos: makeTag(2, "bbb"), lines: ["x", "y", "z"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nx\ny\nz\nbbb"); }); it("inserts before first line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "prepend", pos: makeTag(1, "aaa"), lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "prepend_at", pos: makeTag(1, "aaa"), lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("NEW\naaa\nbbb"); }); it("prepends at BOF without anchor", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "prepend_bof", lines: ["NEW"] }]; + const edits: HashlineEdit[] = [{ op: "prepend_file", lines: ["NEW"] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("NEW\naaa\nbbb"); expect(result.firstChangedLine).toBe(1); @@ -447,7 +447,7 @@ describe("applyHashlineEdits — prepend", () => { it("insert with before and empty text inserts an empty line", () => { const content = "aaa\nbbb"; - const edits: HashlineEdit[] = [{ op: "prepend", pos: makeTag(1, "aaa"), lines: [] }]; + const edits: HashlineEdit[] = [{ op: "prepend_at", pos: makeTag(1, "aaa"), lines: [] }]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("\naaa\nbbb"); expect(result.firstChangedLine).toBe(1); @@ -456,8 +456,8 @@ describe("applyHashlineEdits — prepend", () => { it("insert before and insert after at same line produce correct order", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [ - { op: "prepend", pos: makeTag(2, "bbb"), lines: ["BEFORE"] }, - { op: "append", pos: makeTag(2, "bbb"), lines: ["AFTER"] }, + { op: "prepend_at", pos: makeTag(2, "bbb"), lines: ["BEFORE"] }, + { op: "append_at", pos: makeTag(2, "bbb"), lines: ["AFTER"] }, ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("aaa\nBEFORE\nbbb\nAFTER\nccc"); @@ -466,7 +466,7 @@ describe("applyHashlineEdits — prepend", () => { it("insert before with set at same line", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [ - { op: "prepend", pos: makeTag(2, "bbb"), lines: ["BEFORE"] }, + { op: "prepend_at", pos: makeTag(2, "bbb"), lines: ["BEFORE"] }, { op: "replace_line", pos: makeTag(2, "bbb"), lines: ["BBB"] }, ]; const result = applyHashlineEdits(content, edits); @@ -666,7 +666,7 @@ describe("applyHashlineEdits — multiple edits", () => { const content = "aaa\nbbb\nccc"; const edits: HashlineEdit[] = [ { op: "replace_line", pos: makeTag(3, "ccc"), lines: ["CCC"] }, - { op: "append", pos: makeTag(1, "aaa"), lines: ["INSERTED"] }, + { op: "append_at", pos: makeTag(1, "aaa"), lines: ["INSERTED"] }, ]; const result = applyHashlineEdits(content, edits); @@ -805,10 +805,10 @@ describe("applyHashlineEdits — errors", () => { it("accepts append/prepend with empty text by inserting empty lines", () => { const content = "aaa\nbbb"; - const appendEdits: HashlineEdit[] = [{ op: "append", pos: makeTag(1, "aaa"), lines: [] }]; + const appendEdits: HashlineEdit[] = [{ op: "append_at", pos: makeTag(1, "aaa"), lines: [] }]; expect(applyHashlineEdits(content, appendEdits).lines).toBe("aaa\n\nbbb"); - const prependEdits: HashlineEdit[] = [{ op: "prepend", pos: makeTag(1, "aaa"), lines: [] }]; + const prependEdits: HashlineEdit[] = [{ op: "prepend_at", pos: makeTag(1, "aaa"), lines: [] }]; expect(applyHashlineEdits(content, prependEdits).lines).toBe("\naaa\nbbb"); }); }); From cac628ccceed1cc37d3b0185371f0cddee562e39 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 21:47:48 +0100 Subject: [PATCH 14/22] feat(coding-agent): added git isolation and keybinding utilities for autoresearch - Added git branch isolation for autoresearch sessions with automatic branch creation, reuse, and worktree safety checks. - Added scope definition sections (Files in Scope, Off Limits, Constraints) to autoresearch template for explicit session boundaries. - Added keybinding matcher utilities for consistent escape/cancel key handling across interactive components. - Added ASI metadata validation (hypothesis and rollback context) in log_experiment tool for experiment tracking. - Refactored keybinding logic across 13 components to use centralized matcher functions instead of inline key checks. --- packages/coding-agent/CHANGELOG.md | 19 +++ .../src/autoresearch/command-initialize.md | 3 + .../src/autoresearch/command-resume.md | 4 +- packages/coding-agent/src/autoresearch/git.ts | 149 ++++++++++++++++++ .../coding-agent/src/autoresearch/helpers.ts | 7 + .../coding-agent/src/autoresearch/index.ts | 25 ++- .../coding-agent/src/autoresearch/prompt.md | 12 +- .../src/autoresearch/resume-message.md | 1 + .../src/autoresearch/tools/log-experiment.ts | 40 +++-- .../src/modes/components/agent-dashboard.ts | 9 +- .../extensions/extension-dashboard.ts | 3 +- .../src/modes/components/history-search.ts | 3 +- .../src/modes/components/hook-editor.ts | 3 +- .../src/modes/components/hook-input.ts | 3 +- .../src/modes/components/hook-selector.ts | 3 +- .../src/modes/components/mcp-add-wizard.ts | 3 +- .../src/modes/components/model-selector.ts | 17 +- .../src/modes/components/oauth-selector.ts | 3 +- .../src/modes/components/session-selector.ts | 3 +- .../src/modes/components/settings-selector.ts | 3 +- .../components/status-line-segment-editor.ts | 3 +- .../src/modes/components/tree-selector.ts | 5 +- .../modes/components/user-message-selector.ts | 11 +- .../src/modes/utils/keybinding-matchers.ts | 21 +++ .../test/autoresearch-state.test.ts | 122 +++++++++++++- .../keybindings-escape-components.test.ts | 104 ++++++++++++ 26 files changed, 535 insertions(+), 44 deletions(-) create mode 100644 packages/coding-agent/src/autoresearch/git.ts create mode 100644 packages/coding-agent/src/modes/utils/keybinding-matchers.ts create mode 100644 packages/coding-agent/test/keybindings-escape-components.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bad9eab69..9c9b98b73 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Renamed hashline edit operation types: `append` → `append_at`, `prepend` → `prepend_at`, `append_eof` → `append_file`, `prepend_bof` → `prepend_file` @@ -10,6 +11,12 @@ ### Added +- Added git branch isolation for autoresearch sessions via `ensureAutoresearchBranch()` to safely revert failed experiments +- Added branch status line to autoresearch initialization and resume prompts showing created or reused branch name +- Added `Files in Scope`, `Off Limits`, and `Constraints` sections to autoresearch.md template for explicit scope definition +- Added validation of ASI metadata requirements in `log_experiment` tool, requiring hypothesis for all runs and rollback context for failed runs +- Added keybinding matcher utilities `matchesAppInterrupt()` and `matchesSelectCancel()` for consistent escape key handling across components +- Added support for customizable `app.interrupt` and `tui.select.cancel` keybindings in interactive components - Added `defaultInactive` property to `ToolDefinition` to allow tools to be registered but excluded from the initial active set, with extension responsibility for activation/deactivation - Added dynamic tool activation/deactivation in autoresearch mode via `setActiveTools()` API - Added separate initialization and resume workflows for autoresearch with `command-initialize.md` and `command-resume.md` prompts @@ -36,10 +43,17 @@ ### Changed +- Changed autoresearch startup to create or reuse a dedicated `autoresearch/...` git branch before enabling the experiment loop +- Changed autoresearch to refuse startup when unrelated worktree changes would make auto-reverts unsafe +- Changed autoresearch prompts to emphasize scope and constraints as source of truth for session direction +- Changed component escape key handling to use keybinding manager for `app.interrupt` and `tui.select.cancel` with fallback to raw Escape matching +- Updated autoresearch prompt guidance to require explicit files in scope, off-limits paths, and session constraints - Changed autoresearch command to use intent-based initialization instead of goal parameter, with user input dialog for new sessions +- Changed autoresearch startup to create or reuse a dedicated `autoresearch/...` git branch before enabling the experiment loop, and to refuse startup when unrelated worktree changes would make auto-reverts unsafe - Changed autoresearch startup to activate experiment tools (`init_experiment`, `run_experiment`, `log_experiment`) only when autoresearch mode is enabled - Changed autoresearch shutdown to deactivate experiment tools when mode is disabled or cleared - Changed autoresearch session rehydration to dynamically manage experiment tool activation based on session state +- Changed autoresearch prompts and notes guidance to require explicit files in scope, off-limits paths, and session constraints - Refactored hashline edit validation to enforce stricter anchor requirements per operation type - Updated edit application logic to handle explicit file-level operations (`append_eof`, `prepend_bof`) separately from anchor-based operations - Changed `setWidget` API to accept `ExtensionWidgetOptions` parameter for placement control @@ -62,6 +76,11 @@ - Removed `shouldAutocorrect` function and related boundary line deduplication logic from hashline editor - Removed auto-correction of off-by-one range edits that duplicated closing braces or boundary lines +### Fixed + +- Fixed autoresearch logging to require durable ASI metadata (hypothesis, rollback_reason, next_action_hint) for every run including rollback context for discarded, crashed, and checks-failed experiments +- Fixed autoresearch logging to require durable ASI metadata for every run, including rollback context for discarded, crashed, and checks-failed experiments + ## [13.14.0] - 2026-03-20 ### Added diff --git a/packages/coding-agent/src/autoresearch/command-initialize.md b/packages/coding-agent/src/autoresearch/command-initialize.md index d4791b1d2..271d9135a 100644 --- a/packages/coding-agent/src/autoresearch/command-initialize.md +++ b/packages/coding-agent/src/autoresearch/command-initialize.md @@ -2,10 +2,13 @@ Set up autoresearch for this intent: {{intent}} +{{branch_status_line}} + Explain briefly what autoresearch will do in this repository, then initialize the workspace. Your first actions: - write `autoresearch.md` +- define `Files in Scope`, `Off Limits`, and `Constraints` in `autoresearch.md` - define the benchmark entrypoint in `autoresearch.sh` - optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate - run `init_experiment` diff --git a/packages/coding-agent/src/autoresearch/command-resume.md b/packages/coding-agent/src/autoresearch/command-resume.md index 2ad853a94..46d8cc742 100644 --- a/packages/coding-agent/src/autoresearch/command-resume.md +++ b/packages/coding-agent/src/autoresearch/command-resume.md @@ -2,7 +2,9 @@ Resume autoresearch from the attached notes. @{{autoresearch_md_path}} -Use the notes as the source of truth for the current direction. +{{branch_status_line}} + +Use the notes as the source of truth for the current direction, scope, and constraints. - inspect recent git history for context - inspect `autoresearch.jsonl` if it exists - continue the most promising unfinished branch diff --git a/packages/coding-agent/src/autoresearch/git.ts b/packages/coding-agent/src/autoresearch/git.ts new file mode 100644 index 000000000..12caf3721 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/git.ts @@ -0,0 +1,149 @@ +import type { ExtensionAPI } from "../extensibility/extensions"; +import { PROTECTED_AUTORESEARCH_FILES } from "./helpers"; + +const AUTORESEARCH_BRANCH_PREFIX = "autoresearch/"; +const BRANCH_NAME_MAX_LENGTH = 48; + +export interface EnsureAutoresearchBranchFailure { + error: string; + ok: false; +} + +export interface EnsureAutoresearchBranchSuccess { + branchName: string; + created: boolean; + ok: true; +} + +export type EnsureAutoresearchBranchResult = EnsureAutoresearchBranchFailure | EnsureAutoresearchBranchSuccess; + +export async function ensureAutoresearchBranch( + api: ExtensionAPI, + workDir: string, + goal: string | null, +): Promise { + const repoRootResult = await api.exec("git", ["rev-parse", "--show-toplevel"], { cwd: workDir, timeout: 5_000 }); + if (repoRootResult.code !== 0) { + return { + error: "Autoresearch requires a git repository so it can isolate experiments and revert failed runs safely.", + ok: false, + }; + } + + const currentBranchResult = await api.exec("git", ["branch", "--show-current"], { cwd: workDir, timeout: 5_000 }); + const currentBranch = currentBranchResult.stdout.trim(); + if (currentBranch.startsWith(AUTORESEARCH_BRANCH_PREFIX)) { + return { + branchName: currentBranch, + created: false, + ok: true, + }; + } + + const dirtyPathsResult = await api.exec("git", ["status", "--porcelain", "--untracked-files=all"], { + cwd: workDir, + timeout: 5_000, + }); + if (dirtyPathsResult.code !== 0) { + return { + error: `Unable to inspect git status before starting autoresearch: ${mergeStdoutStderr(dirtyPathsResult).trim() || `exit ${dirtyPathsResult.code}`}`, + ok: false, + }; + } + + const unsafeDirtyPaths = parseUnsafeDirtyPaths(dirtyPathsResult.stdout); + if (unsafeDirtyPaths.length > 0) { + const preview = unsafeDirtyPaths.slice(0, 5).join(", "); + const suffix = unsafeDirtyPaths.length > 5 ? ` (+${unsafeDirtyPaths.length - 5} more)` : ""; + return { + error: + "Autoresearch needs a clean git worktree before it can create an isolated branch. " + + `Commit or stash these paths first: ${preview}${suffix}`, + ok: false, + }; + } + + const branchName = await allocateBranchName(api, workDir, goal); + const checkoutResult = await api.exec("git", ["checkout", "-b", branchName], { cwd: workDir, timeout: 10_000 }); + if (checkoutResult.code !== 0) { + return { + error: + `Failed to create autoresearch branch ${branchName}: ` + + `${mergeStdoutStderr(checkoutResult).trim() || `exit ${checkoutResult.code}`}`, + ok: false, + }; + } + + return { + branchName, + created: true, + ok: true, + }; +} + +function parseUnsafeDirtyPaths(statusOutput: string): string[] { + const unsafePaths = new Set(); + for (const line of statusOutput.split("\n")) { + const trimmedLine = line.trimEnd(); + if (trimmedLine.length < 4) continue; + const rawPath = trimmedLine.slice(3).trim(); + if (rawPath.length === 0) continue; + const renameParts = rawPath.split(" -> "); + const normalizedPath = normalizeStatusPath(renameParts[renameParts.length - 1] ?? rawPath); + if (normalizedPath.length === 0) continue; + if (PROTECTED_AUTORESEARCH_FILES.some(path => path === normalizedPath)) continue; + unsafePaths.add(normalizedPath); + } + return [...unsafePaths]; +} + +function normalizeStatusPath(path: string): string { + let normalized = path.trim(); + if (normalized.startsWith('"') && normalized.endsWith('"')) { + normalized = normalized.slice(1, -1); + } + if (normalized.startsWith("./")) { + normalized = normalized.slice(2); + } + return normalized; +} + +async function allocateBranchName(api: ExtensionAPI, workDir: string, goal: string | null): Promise { + const baseName = `${AUTORESEARCH_BRANCH_PREFIX}${slugifyGoal(goal)}-${currentDateStamp()}`; + let candidate = baseName; + let suffix = 2; + while (await branchExists(api, workDir, candidate)) { + candidate = `${baseName}-${suffix}`; + suffix += 1; + } + return candidate; +} + +async function branchExists(api: ExtensionAPI, workDir: string, branchName: string): Promise { + const result = await api.exec("git", ["show-ref", "--verify", "--quiet", `refs/heads/${branchName}`], { + cwd: workDir, + timeout: 5_000, + }); + return result.code === 0; +} + +function slugifyGoal(goal: string | null): string { + const normalized = (goal ?? "") + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, ""); + const trimmed = normalized.slice(0, BRANCH_NAME_MAX_LENGTH).replace(/-+$/g, ""); + return trimmed || "session"; +} + +function currentDateStamp(): string { + const now = new Date(); + const year = String(now.getFullYear()); + const month = String(now.getMonth() + 1).padStart(2, "0"); + const day = String(now.getDate()).padStart(2, "0"); + return `${year}${month}${day}`; +} + +function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { + return `${result.stdout}${result.stderr}`; +} diff --git a/packages/coding-agent/src/autoresearch/helpers.ts b/packages/coding-agent/src/autoresearch/helpers.ts index d5b845ef7..681c50f41 100644 --- a/packages/coding-agent/src/autoresearch/helpers.ts +++ b/packages/coding-agent/src/autoresearch/helpers.ts @@ -9,6 +9,13 @@ export const METRIC_LINE_PREFIX = "METRIC"; export const ASI_LINE_PREFIX = "ASI"; export const EXPERIMENT_MAX_LINES = 10; export const EXPERIMENT_MAX_BYTES = 4 * 1024; +export const PROTECTED_AUTORESEARCH_FILES = [ + "autoresearch.jsonl", + "autoresearch.md", + "autoresearch.ideas.md", + "autoresearch.sh", + "autoresearch.checks.sh", +] as const; const DENIED_KEY_NAMES = new Set(["__proto__", "constructor", "prototype"]); diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 8f13e5cdc..6de676653 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -6,6 +6,7 @@ import type { ExtensionContext, ExtensionFactory } from "../extensibility/extens import commandInitializeTemplate from "./command-initialize.md" with { type: "text" }; import commandResumeTemplate from "./command-resume.md" with { type: "text" }; import { createDashboardController } from "./dashboard"; +import { ensureAutoresearchBranch } from "./git"; import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "./helpers"; import promptTemplate from "./prompt.md" with { type: "text" }; import resumeMessageTemplate from "./resume-message.md" with { type: "text" }; @@ -131,6 +132,12 @@ export const createAutoresearchExtension: ExtensionFactory = api => { const hasAutoresearchMd = fs.existsSync(autoresearchMdPath); if (hasAutoresearchMd) { + const branchResult = await ensureAutoresearchBranch(api, workDir, runtime.goal); + if (!branchResult.ok) { + ctx.ui.notify(branchResult.error, "error"); + return; + } + setMode(ctx, true, runtime.goal, "on"); runtime.experimentsThisSession = 0; runtime.autoResumeTurns = 0; @@ -139,6 +146,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => { api.sendUserMessage( renderPromptTemplate(commandResumeTemplate, { autoresearch_md_path: autoresearchMdPath, + branch_status_line: branchResult.created + ? `Created and checked out dedicated git branch \`${branchResult.branchName}\` before resuming.` + : `Using dedicated git branch \`${branchResult.branchName}\`.`, }), ); return; @@ -156,12 +166,25 @@ export const createAutoresearchExtension: ExtensionFactory = api => { return; } + const branchResult = await ensureAutoresearchBranch(api, workDir, intent); + if (!branchResult.ok) { + ctx.ui.notify(branchResult.error, "error"); + return; + } + setMode(ctx, true, intent, "on"); runtime.experimentsThisSession = 0; runtime.autoResumeTurns = 0; dashboard.updateWidget(ctx, runtime); await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); - api.sendUserMessage(renderPromptTemplate(commandInitializeTemplate, { intent })); + api.sendUserMessage( + renderPromptTemplate(commandInitializeTemplate, { + branch_status_line: branchResult.created + ? `Created and checked out dedicated git branch \`${branchResult.branchName}\`.` + : `Using dedicated git branch \`${branchResult.branchName}\`.`, + intent, + }), + ); }, }); diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md index 56ce3efd3..2ed34f189 100644 --- a/packages/coding-agent/src/autoresearch/prompt.md +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -25,7 +25,7 @@ You are running an autonomous experiment loop. Keep iterating until the user int - Identify the true bottleneck or quality constraint. - Check existing scripts, benchmark harnesses, and config files. 2. Keep your notes in `autoresearch.md`. - - Record the goal, the benchmark command, the primary metric, important secondary metrics, and the running ideas backlog. + - Record the goal, the benchmark command, the primary metric, important secondary metrics, the files in scope, hard constraints, and the running ideas backlog. - Update the notes whenever the strategy changes. 3. Use `autoresearch.sh` as the canonical benchmark entrypoint. - If it does not exist yet, create it. @@ -87,6 +87,15 @@ Suggested structure: - primary metric: - secondary metrics: +## Files in Scope +- path: + +## Off Limits +- path: + +## Constraints +- rule: + ## Baseline - metric: - notes: @@ -104,6 +113,7 @@ Suggested structure: - Do not game the benchmark. - Do not overfit to synthetic inputs if the real workload is broader. - Preserve correctness. +- Only modify files that are explicitly in scope for the current session. - If you create `autoresearch.checks.sh`, treat it as a hard gate for `keep`. - If the user sends another message while a run is in progress, finish the current run and logging cycle first, then address the new input in the next iteration. diff --git a/packages/coding-agent/src/autoresearch/resume-message.md b/packages/coding-agent/src/autoresearch/resume-message.md index 55fda95ad..62c10b26a 100644 --- a/packages/coding-agent/src/autoresearch/resume-message.md +++ b/packages/coding-agent/src/autoresearch/resume-message.md @@ -1,6 +1,7 @@ The autoresearch loop ended unexpectedly. Resume it now. - Read `autoresearch.md` and `autoresearch.jsonl`. +- Treat `autoresearch.md` as the source of truth for the current direction, scope, and constraints. - Inspect recent git history for context. - Continue from the most promising unfinished direction. {{#if has_ideas}} diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index e0dfa0e53..2a7420a46 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -5,7 +5,14 @@ import { Text } from "@oh-my-pi/pi-tui"; import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; -import { formatNum, inferMetricUnitFromName, mergeAsi, resolveWorkDir, validateWorkDir } from "../helpers"; +import { + formatNum, + inferMetricUnitFromName, + mergeAsi, + PROTECTED_AUTORESEARCH_FILES, + resolveWorkDir, + validateWorkDir, +} from "../helpers"; import { cloneExperimentState, computeConfidence, @@ -52,14 +59,6 @@ const logExperimentSchema = Type.Object({ ), }); -const PROTECTED_AUTORESEARCH_FILES = [ - "autoresearch.jsonl", - "autoresearch.md", - "autoresearch.ideas.md", - "autoresearch.sh", - "autoresearch.checks.sh", -] as const; - interface PreservedFile { content: Buffer; path: string; @@ -107,6 +106,12 @@ export function createLogExperimentTool( } const mergedAsi = mergeAsi(runtime.lastRunAsi, sanitizeAsi(params.asi)); + const asiValidationError = validateAsiRequirements(mergedAsi, params.status); + if (asiValidationError) { + return { + content: [{ type: "text", text: `Error: ${asiValidationError}` }], + }; + } const experiment: ExperimentResult = { commit: params.commit.slice(0, 7), metric: params.metric, @@ -220,6 +225,23 @@ function sanitizeAsiValue(value: unknown): ASIData[string] | undefined { return undefined; } +export function validateAsiRequirements(asi: ASIData | undefined, status: ExperimentResult["status"]): string | null { + if (!asi) { + return "asi is required. Include at minimum a non-empty hypothesis."; + } + if (typeof asi.hypothesis !== "string" || asi.hypothesis.trim().length === 0) { + return "asi.hypothesis is required and must be a non-empty string."; + } + if (status === "keep") return null; + if (typeof asi.rollback_reason !== "string" || asi.rollback_reason.trim().length === 0) { + return "asi.rollback_reason is required for discard, crash, and checks_failed results."; + } + if (typeof asi.next_action_hint !== "string" || asi.next_action_hint.trim().length === 0) { + return "asi.next_action_hint is required for discard, crash, and checks_failed results."; + } + return null; +} + function validateSecondaryMetrics(state: ExperimentState, metrics: NumericMetricMap, force: boolean): string | null { if (state.secondaryMetrics.length === 0) return null; const knownNames = new Set(state.secondaryMetrics.map(metric => metric.name)); diff --git a/packages/coding-agent/src/modes/components/agent-dashboard.ts b/packages/coding-agent/src/modes/components/agent-dashboard.ts index cd5f46d32..ade3e3119 100644 --- a/packages/coding-agent/src/modes/components/agent-dashboard.ts +++ b/packages/coding-agent/src/modes/components/agent-dashboard.ts @@ -50,6 +50,7 @@ import { discoverAgents } from "../../task/discovery"; import type { AgentDefinition, AgentSource } from "../../task/types"; import { shortenPath } from "../../tools/render-utils"; import { theme } from "../theme/theme"; +import { matchesAppInterrupt } from "../utils/keybinding-matchers"; import { DynamicBorder } from "./dynamic-border"; type SourceTabId = "all" | AgentSource; @@ -993,7 +994,7 @@ export class AgentDashboard extends Container { } if (this.#createSpec) { - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (matchesAppInterrupt(data)) { this.#clearCreateFlow(); this.#buildLayout(); return; @@ -1017,7 +1018,7 @@ export class AgentDashboard extends Container { } if (this.#createInput || this.#createGenerating) { - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (matchesAppInterrupt(data)) { if (!this.#createGenerating) { this.#clearCreateFlow(); this.#buildLayout(); @@ -1037,7 +1038,7 @@ export class AgentDashboard extends Container { } if (this.#editInput) { - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (matchesAppInterrupt(data)) { this.#cancelModelEdit(); return; } @@ -1048,7 +1049,7 @@ export class AgentDashboard extends Container { return; } - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (matchesAppInterrupt(data)) { if (this.#searchQuery.length > 0) { this.#searchQuery = ""; this.#applyFilters(); diff --git a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts index 26b3d7217..890611a9f 100644 --- a/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts +++ b/packages/coding-agent/src/modes/components/extensions/extension-dashboard.ts @@ -24,6 +24,7 @@ import { import { Settings } from "../../../config/settings"; import { DynamicBorder } from "../../../modes/components/dynamic-border"; import { theme } from "../../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../../modes/utils/keybinding-matchers"; import { ExtensionList } from "./extension-list"; import { InspectorPanel } from "./inspector-panel"; import { applyFilter, createInitialState, filterByProvider, refreshState, toggleProvider } from "./state-manager"; @@ -251,7 +252,7 @@ export class ExtensionDashboard extends Container { } // Escape - clear search first, then close - if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + if (matchesAppInterrupt(data)) { if (this.#state.searchQuery.length > 0) { this.#state.searchQuery = ""; this.#state.searchFiltered = this.#state.tabFiltered; diff --git a/packages/coding-agent/src/modes/components/history-search.ts b/packages/coding-agent/src/modes/components/history-search.ts index 34d741a2c..e3814eef5 100644 --- a/packages/coding-agent/src/modes/components/history-search.ts +++ b/packages/coding-agent/src/modes/components/history-search.ts @@ -11,6 +11,7 @@ import { visibleWidth, } from "@oh-my-pi/pi-tui"; import { theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import type { HistoryEntry, HistoryStorage } from "../../session/history-storage"; import { DynamicBorder } from "./dynamic-border"; @@ -137,7 +138,7 @@ export class HistorySearchComponent extends Container { return; } - if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + if (matchesAppInterrupt(keyData)) { this.#onCancel(); return; } diff --git a/packages/coding-agent/src/modes/components/hook-editor.ts b/packages/coding-agent/src/modes/components/hook-editor.ts index 56d9fce29..94ddfd62b 100644 --- a/packages/coding-agent/src/modes/components/hook-editor.ts +++ b/packages/coding-agent/src/modes/components/hook-editor.ts @@ -4,6 +4,7 @@ */ import { Container, Editor, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; import { getEditorTheme, theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import { getEditorCommand, openInEditor } from "../../utils/external-editor"; import { DynamicBorder } from "./dynamic-border"; @@ -67,7 +68,7 @@ export class HookEditorComponent extends Container { } // Escape to cancel - if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + if (matchesAppInterrupt(keyData)) { this.#onCancelCallback(); return; } diff --git a/packages/coding-agent/src/modes/components/hook-input.ts b/packages/coding-agent/src/modes/components/hook-input.ts index eb4d01f66..7a42ecde1 100644 --- a/packages/coding-agent/src/modes/components/hook-input.ts +++ b/packages/coding-agent/src/modes/components/hook-input.ts @@ -3,6 +3,7 @@ */ import { Container, Input, Markdown, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; import { getMarkdownTheme, theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import { CountdownTimer } from "./countdown-timer"; import { DynamicBorder } from "./dynamic-border"; @@ -65,7 +66,7 @@ export class HookInputComponent extends Container { this.#countdown?.reset(); if (matchesKey(keyData, "enter") || matchesKey(keyData, "return") || keyData === "\n") { this.#onSubmitCallback(this.#input.getValue()); - } else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + } else if (matchesAppInterrupt(keyData)) { this.#onCancelCallback(); } else { this.#input.handleInput(keyData); diff --git a/packages/coding-agent/src/modes/components/hook-selector.ts b/packages/coding-agent/src/modes/components/hook-selector.ts index c00aea1d1..d39e973b7 100644 --- a/packages/coding-agent/src/modes/components/hook-selector.ts +++ b/packages/coding-agent/src/modes/components/hook-selector.ts @@ -16,6 +16,7 @@ import { visibleWidth, } from "@oh-my-pi/pi-tui"; import { getMarkdownTheme, theme } from "../../modes/theme/theme"; +import { matchesSelectCancel } from "../../modes/utils/keybinding-matchers"; import { CountdownTimer } from "./countdown-timer"; import { DynamicBorder } from "./dynamic-border"; @@ -173,7 +174,7 @@ export class HookSelectorComponent extends Container { this.#onLeftCallback?.(); } else if (matchesKey(keyData, "right")) { this.#onRightCallback?.(); - } else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc") || matchesKey(keyData, "ctrl+c")) { + } else if (matchesSelectCancel(keyData)) { this.#onCancelCallback(); } } diff --git a/packages/coding-agent/src/modes/components/mcp-add-wizard.ts b/packages/coding-agent/src/modes/components/mcp-add-wizard.ts index 1e6fc22a2..88a3dc199 100644 --- a/packages/coding-agent/src/modes/components/mcp-add-wizard.ts +++ b/packages/coding-agent/src/modes/components/mcp-add-wizard.ts @@ -19,6 +19,7 @@ import { analyzeAuthError, discoverOAuthEndpoints } from "../../mcp/oauth-discov import type { MCPHttpServerConfig, MCPServerConfig, MCPSseServerConfig, MCPStdioServerConfig } from "../../mcp/types"; import { shortenPath } from "../../tools/render-utils"; import { theme } from "../theme/theme"; +import { matchesAppInterrupt } from "../utils/keybinding-matchers"; import { DynamicBorder } from "./dynamic-border"; type TransportType = "stdio" | "http" | "sse"; @@ -452,7 +453,7 @@ export class MCPAddWizard extends Container { } // Handle Escape (always handled by wizard) - if (matchesKey(keyData, "escape")) { + if (matchesAppInterrupt(keyData)) { if (this.#currentStep === "name") { // Cancel wizard this.#onCancelCallback(); diff --git a/packages/coding-agent/src/modes/components/model-selector.ts b/packages/coding-agent/src/modes/components/model-selector.ts index 75e6c8e44..2108ce48d 100644 --- a/packages/coding-agent/src/modes/components/model-selector.ts +++ b/packages/coding-agent/src/modes/components/model-selector.ts @@ -1,6 +1,17 @@ import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { getSupportedEfforts, type Model, modelsAreEqual } from "@oh-my-pi/pi-ai"; -import { Container, Input, matchesKey, Spacer, type Tab, TabBar, Text, type TUI, visibleWidth } from "@oh-my-pi/pi-tui"; +import { + Container, + getKeybindings, + Input, + matchesKey, + Spacer, + type Tab, + TabBar, + Text, + type TUI, + visibleWidth, +} from "@oh-my-pi/pi-tui"; import { MODEL_ROLE_IDS, MODEL_ROLES, type ModelRegistry, type ModelRole } from "../../config/model-registry"; import { resolveModelRoleValue } from "../../config/model-resolver"; import type { Settings } from "../../config/settings"; @@ -647,7 +658,7 @@ export class ModelSelectorComponent extends Container { } // Escape or Ctrl+C - close selector - if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc") || matchesKey(keyData, "ctrl+c")) { + if (getKeybindings().matches(keyData, "tui.select.cancel")) { this.#onCancelCallback(); return; } @@ -698,7 +709,7 @@ export class ModelSelectorComponent extends Container { return; } - if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc") || matchesKey(keyData, "ctrl+c")) { + if (getKeybindings().matches(keyData, "tui.select.cancel")) { if (this.#menuStep === "thinking" && this.#menuSelectedRole !== null) { this.#menuStep = "role"; const roleIndex = MENU_ROLE_ACTIONS.findIndex(action => action.role === this.#menuSelectedRole); diff --git a/packages/coding-agent/src/modes/components/oauth-selector.ts b/packages/coding-agent/src/modes/components/oauth-selector.ts index 6602805a9..edd6eb49a 100644 --- a/packages/coding-agent/src/modes/components/oauth-selector.ts +++ b/packages/coding-agent/src/modes/components/oauth-selector.ts @@ -1,6 +1,7 @@ import { getOAuthProviders, type OAuthProviderInfo } from "@oh-my-pi/pi-ai"; import { Container, matchesKey, Spacer, TruncatedText } from "@oh-my-pi/pi-tui"; import { theme } from "../../modes/theme/theme"; +import { matchesSelectCancel } from "../../modes/utils/keybinding-matchers"; import type { AuthStorage } from "../../session/auth-storage"; import { DynamicBorder } from "./dynamic-border"; /** @@ -202,7 +203,7 @@ export class OAuthSelectorComponent extends Container { } } // Escape or Ctrl+C - else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc") || matchesKey(keyData, "ctrl+c")) { + else if (matchesSelectCancel(keyData)) { this.stopValidation(); this.#onCancelCallback(); } diff --git a/packages/coding-agent/src/modes/components/session-selector.ts b/packages/coding-agent/src/modes/components/session-selector.ts index 2739cc0ee..8087fcb7c 100644 --- a/packages/coding-agent/src/modes/components/session-selector.ts +++ b/packages/coding-agent/src/modes/components/session-selector.ts @@ -11,6 +11,7 @@ import { visibleWidth, } from "@oh-my-pi/pi-tui"; import { theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import type { SessionInfo } from "../../session/session-manager"; import { fuzzyFilter } from "../../utils/fuzzy"; import { DynamicBorder } from "./dynamic-border"; @@ -219,7 +220,7 @@ class SessionList implements Component { return; } // Escape - cancel - if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + if (matchesAppInterrupt(keyData)) { if (this.onCancel) { this.onCancel(); } diff --git a/packages/coding-agent/src/modes/components/settings-selector.ts b/packages/coding-agent/src/modes/components/settings-selector.ts index 930bcefda..f6cc76ea7 100644 --- a/packages/coding-agent/src/modes/components/settings-selector.ts +++ b/packages/coding-agent/src/modes/components/settings-selector.ts @@ -21,6 +21,7 @@ import type { } from "../../config/settings-schema"; import { SETTING_TABS, TAB_METADATA } from "../../config/settings-schema"; import { getCurrentThemeName, getSelectListTheme, getSettingsListTheme, theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import { getTabBarTheme } from "../shared"; import { DynamicBorder } from "./dynamic-border"; import { PluginSettingsComponent } from "./plugin-settings"; @@ -521,7 +522,7 @@ export class SettingsSelectorComponent extends Container { } // Escape at top level cancels - if ((matchesKey(data, "escape") || matchesKey(data, "esc")) && !this.#currentSubmenu) { + if (matchesAppInterrupt(data) && !this.#currentSubmenu) { this.callbacks.onCancel(); return; } diff --git a/packages/coding-agent/src/modes/components/status-line-segment-editor.ts b/packages/coding-agent/src/modes/components/status-line-segment-editor.ts index c2f7dcf0d..f30050c8f 100644 --- a/packages/coding-agent/src/modes/components/status-line-segment-editor.ts +++ b/packages/coding-agent/src/modes/components/status-line-segment-editor.ts @@ -11,6 +11,7 @@ import { Container, matchesKey, padding } from "@oh-my-pi/pi-tui"; import type { StatusLineSegmentId } from "../../config/settings-schema"; import { theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import { ALL_SEGMENT_IDS } from "./status-line/segments"; // Segment display names and short descriptions @@ -239,7 +240,7 @@ export class StatusLineSegmentEditorComponent extends Container { const left = this.#getSegmentsForColumn("left").map(s => s.id); const right = this.#getSegmentsForColumn("right").map(s => s.id); this.callbacks.onSave(left, right); - } else if (matchesKey(data, "escape") || matchesKey(data, "esc")) { + } else if (matchesAppInterrupt(data)) { this.callbacks.onCancel(); } } diff --git a/packages/coding-agent/src/modes/components/tree-selector.ts b/packages/coding-agent/src/modes/components/tree-selector.ts index a2491bc3a..19649ecb4 100644 --- a/packages/coding-agent/src/modes/components/tree-selector.ts +++ b/packages/coding-agent/src/modes/components/tree-selector.ts @@ -12,6 +12,7 @@ import { } from "@oh-my-pi/pi-tui"; import type { TreeFilterMode } from "../../config/settings-schema"; import { theme } from "../../modes/theme/theme"; +import { matchesAppInterrupt } from "../../modes/utils/keybinding-matchers"; import type { SessionTreeNode } from "../../session/session-manager"; import { shortenPath } from "../../tools/render-utils"; import { DynamicBorder } from "./dynamic-border"; @@ -702,7 +703,7 @@ class TreeList implements Component { if (selected && this.onSelect) { this.onSelect(selected.node.entry.id); } - } else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + } else if (matchesAppInterrupt(keyData)) { if (this.#searchQuery) { this.#searchQuery = ""; this.#applyFilter(); @@ -807,7 +808,7 @@ class LabelInput implements Component { if (matchesKey(keyData, "enter") || matchesKey(keyData, "return") || keyData === "\n") { const value = this.#input.getValue().trim(); this.onSubmit?.(this.entryId, value || undefined); - } else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { + } else if (matchesAppInterrupt(keyData)) { this.onCancel?.(); } else { this.#input.handleInput(keyData); diff --git a/packages/coding-agent/src/modes/components/user-message-selector.ts b/packages/coding-agent/src/modes/components/user-message-selector.ts index 7b6dffff9..2d5066208 100644 --- a/packages/coding-agent/src/modes/components/user-message-selector.ts +++ b/packages/coding-agent/src/modes/components/user-message-selector.ts @@ -1,5 +1,6 @@ import { type Component, Container, matchesKey, Spacer, Text, truncateToWidth } from "@oh-my-pi/pi-tui"; import { theme } from "../../modes/theme/theme"; +import { matchesSelectCancel } from "../../modes/utils/keybinding-matchers"; import { DynamicBorder } from "./dynamic-border"; interface UserMessageItem { @@ -91,14 +92,8 @@ class UserMessageList implements Component { this.onSelect(selected.id); } } - // Escape - cancel - else if (matchesKey(keyData, "escape") || matchesKey(keyData, "esc")) { - if (this.onCancel) { - this.onCancel(); - } - } - // Ctrl+C - cancel - else if (matchesKey(keyData, "ctrl+c")) { + // Escape / cancel + else if (matchesSelectCancel(keyData)) { if (this.onCancel) { this.onCancel(); } diff --git a/packages/coding-agent/src/modes/utils/keybinding-matchers.ts b/packages/coding-agent/src/modes/utils/keybinding-matchers.ts new file mode 100644 index 000000000..7fc7383f6 --- /dev/null +++ b/packages/coding-agent/src/modes/utils/keybinding-matchers.ts @@ -0,0 +1,21 @@ +import { getKeybindings, matchesKey } from "@oh-my-pi/pi-tui"; + +/** + * Match the coding-agent interrupt key. + * + * Interactive mode installs a keybinding manager that exposes `app.interrupt` + * globally, but some isolated component tests still run with only TUI + * keybindings registered. In that case, fall back to raw Escape matching. + */ +export function matchesAppInterrupt(data: string): boolean { + const keybindings = getKeybindings(); + const interruptKeys = keybindings.getKeys("app.interrupt"); + if (interruptKeys.length > 0) { + return keybindings.matches(data, "app.interrupt"); + } + return matchesKey(data, "escape") || matchesKey(data, "esc"); +} + +export function matchesSelectCancel(data: string): boolean { + return getKeybindings().matches(data, "tui.select.cancel"); +} diff --git a/packages/coding-agent/test/autoresearch-state.test.ts b/packages/coding-agent/test/autoresearch-state.test.ts index c97be3d11..285cd70da 100644 --- a/packages/coding-agent/test/autoresearch-state.test.ts +++ b/packages/coding-agent/test/autoresearch-state.test.ts @@ -6,6 +6,7 @@ import { Snowflake } from "@oh-my-pi/pi-utils"; import { isAutoresearchShCommand } from "../src/autoresearch/helpers"; import { createAutoresearchExtension } from "../src/autoresearch/index"; import { reconstructStateFromJsonl } from "../src/autoresearch/state"; +import { validateAsiRequirements } from "../src/autoresearch/tools/log-experiment"; import type { ExtensionAPI, ExtensionCommandContext, @@ -118,12 +119,18 @@ describe("autoresearch command guard", () => { interface AutoresearchCommandHarness { command: RegisteredCommand; ctx: ExtensionCommandContext; + execCalls: Array<{ args: string[]; command: string }>; sentMessages: string[]; inputCalls: Array<{ title: string; placeholder: string | undefined }>; notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }>; } -function createAutoresearchCommandHarness(cwd: string, inputResult: string | undefined): AutoresearchCommandHarness { +function createAutoresearchCommandHarness( + cwd: string, + inputResult: string | undefined, + execImpl?: (command: string, args: string[]) => Promise<{ code: number; stderr: string; stdout: string }>, +): AutoresearchCommandHarness { + const execCalls: Array<{ args: string[]; command: string }> = []; const sentMessages: string[] = []; const inputCalls: Array<{ title: string; placeholder: string | undefined }> = []; const notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }> = []; @@ -131,6 +138,13 @@ function createAutoresearchCommandHarness(cwd: string, inputResult: string | und const api = { appendEntry(_customType: string, _data?: unknown): void {}, + exec: async (commandName: string, args: string[]) => { + execCalls.push({ args: [...args], command: commandName }); + if (execImpl) { + return execImpl(commandName, args); + } + return { code: 0, stderr: "", stdout: "" }; + }, on(): void {}, registerCommand(name: string, options: Omit): void { command = { name, ...options }; @@ -191,7 +205,7 @@ function createAutoresearchCommandHarness(cwd: string, inputResult: string | und waitForIdle: async () => {}, } as unknown as ExtensionCommandContext; - return { command, ctx, sentMessages, inputCalls, notifications }; + return { command, ctx, execCalls, sentMessages, inputCalls, notifications }; } interface AutoresearchLifecycleHarness { @@ -289,7 +303,30 @@ describe("autoresearch command startup", () => { it("asks for intent and sends an initialization prompt when no autoresearch.md exists", async () => { const dir = makeTempDir(); tempDirs.push(dir); - const harness = createAutoresearchCommandHarness(dir, "reduce edit benchmark runtime variance"); + let currentBranch = "main"; + const branches = new Set(); + const harness = createAutoresearchCommandHarness( + dir, + "reduce edit benchmark runtime variance", + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: `${currentBranch}\n` }; + } + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; + if (args[0] === "show-ref") { + const branchName = args[args.length - 1]?.replace("refs/heads/", "") ?? ""; + return { code: branches.has(branchName) ? 0 : 1, stderr: "", stdout: "" }; + } + if (args[0] === "checkout" && args[1] === "-b") { + currentBranch = args[2] ?? currentBranch; + branches.add(currentBranch); + return { code: 0, stderr: "", stdout: "" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); await harness.command.handler("", harness.ctx); @@ -299,8 +336,12 @@ describe("autoresearch command startup", () => { expect(harness.sentMessages).toHaveLength(1); expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); expect(harness.sentMessages[0]).toContain("reduce edit benchmark runtime variance"); + expect(harness.sentMessages[0]).toContain("Created and checked out dedicated git branch"); expect(harness.sentMessages[0]).toContain("Explain briefly what autoresearch will do in this repository"); + expect(harness.sentMessages[0]).toContain("Files in Scope"); expect(harness.notifications).toEqual([]); + const checkoutCall = harness.execCalls.find(call => call.command === "git" && call.args[0] === "checkout"); + expect(checkoutCall?.args[2]).toMatch(/^autoresearch\/reduce-edit-benchmark-runtime-variance-\d{8}$/); }); it("resumes from autoresearch.md without asking for intent when notes already exist", async () => { @@ -308,7 +349,14 @@ describe("autoresearch command startup", () => { tempDirs.push(dir); const autoresearchMdPath = path.join(dir, "autoresearch.md"); fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); - const harness = createAutoresearchCommandHarness(dir, "ignored"); + const harness = createAutoresearchCommandHarness(dir, "ignored", async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }); await harness.command.handler("", harness.ctx); @@ -319,7 +367,9 @@ describe("autoresearch command startup", () => { "", `@${autoresearchMdPath}`, "", - "Use the notes as the source of truth for the current direction.", + "Using dedicated git branch `autoresearch/existing-20260322`.", + "", + "Use the notes as the source of truth for the current direction, scope, and constraints.", "- inspect recent git history for context", "- inspect `autoresearch.jsonl` if it exists", "- continue the most promising unfinished branch", @@ -338,6 +388,37 @@ describe("autoresearch command startup", () => { expect(harness.sentMessages).toEqual([]); expect(harness.notifications).toEqual([{ message: "Autoresearch intent is required", type: "info" }]); }); + + it("refuses to start when non-autoresearch files are dirty on a non-autoresearch branch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const harness = createAutoresearchCommandHarness( + dir, + "reduce edit benchmark runtime variance", + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "main\n" }; + } + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: " M packages/coding-agent/src/sdk.ts\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([ + { + message: + "Autoresearch needs a clean git worktree before it can create an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", + type: "error", + }, + ]); + }); }); describe("autoresearch lifecycle tool activation", () => { @@ -369,3 +450,34 @@ describe("autoresearch lifecycle tool activation", () => { expect(harness.setActiveToolsCalls).toEqual([["read"]]); }); }); + +describe("autoresearch ASI requirements", () => { + it("requires a hypothesis for every run", () => { + expect(validateAsiRequirements(undefined, "keep")).toBe( + "asi is required. Include at minimum a non-empty hypothesis.", + ); + expect(validateAsiRequirements({}, "keep")).toBe("asi.hypothesis is required and must be a non-empty string."); + }); + + it("requires rollback metadata for failed runs", () => { + expect(validateAsiRequirements({ hypothesis: "try a smaller cache" }, "discard")).toBe( + "asi.rollback_reason is required for discard, crash, and checks_failed results.", + ); + expect( + validateAsiRequirements( + { hypothesis: "try a smaller cache", rollback_reason: "metric regressed" }, + "checks_failed", + ), + ).toBe("asi.next_action_hint is required for discard, crash, and checks_failed results."); + expect( + validateAsiRequirements( + { + hypothesis: "try a smaller cache", + next_action_hint: "re-run with lower batch size", + rollback_reason: "metric regressed", + }, + "crash", + ), + ).toBeNull(); + }); +}); diff --git a/packages/coding-agent/test/keybindings-escape-components.test.ts b/packages/coding-agent/test/keybindings-escape-components.test.ts new file mode 100644 index 000000000..515091473 --- /dev/null +++ b/packages/coding-agent/test/keybindings-escape-components.test.ts @@ -0,0 +1,104 @@ +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import { getBundledModel } from "@oh-my-pi/pi-ai"; +import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; +import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { ModelSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/model-selector"; +import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { setKeybindings, type TUI } from "@oh-my-pi/pi-tui"; + +beforeAll(() => { + initTheme(); +}); + +afterEach(() => { + setKeybindings(KeybindingsManager.inMemory()); + vi.restoreAllMocks(); +}); + +function createSession(id: string, title: string): SessionInfo { + return { + path: `/tmp/${id}.jsonl`, + id, + cwd: "/tmp", + title, + created: new Date("2024-01-01T00:00:00Z"), + modified: new Date("2024-01-02T00:00:00Z"), + messageCount: 1, + firstMessage: `${title} first message`, + allMessagesText: `${title} first message`, + }; +} + +describe("component escape bindings", () => { + it("uses app.interrupt for session selector cancel without changing Ctrl+C exit", () => { + const keybindings = KeybindingsManager.inMemory({ + "app.interrupt": "alt+x", + }); + setKeybindings(keybindings); + + const onCancel = vi.fn(); + const onExit = vi.fn(); + const selector = new SessionSelectorComponent( + [createSession("session-a", "Alpha"), createSession("session-b", "Beta")], + () => {}, + onCancel, + onExit, + ); + + selector.handleInput("\x1b"); + expect(onCancel).not.toHaveBeenCalled(); + + selector.handleInput("\x1bx"); + expect(onCancel).toHaveBeenCalledTimes(1); + + selector.handleInput("\x03"); + expect(onExit).toHaveBeenCalledTimes(1); + }); + + it("uses tui.select.cancel for model selector cancellation", async () => { + const keybindings = KeybindingsManager.inMemory({ + "tui.select.cancel": "ctrl+g", + }); + setKeybindings(keybindings); + + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) { + throw new Error("Expected bundled model anthropic/claude-sonnet-4-5"); + } + + const settings = Settings.isolated({ + modelRoles: { + default: `${model.provider}/${model.id}`, + }, + }); + const modelRegistry = { + getAll: () => [model], + getDiscoverableProviders: () => [], + } as unknown as ModelRegistry; + const ui = { + requestRender: vi.fn(), + } as unknown as TUI; + const onCancel = vi.fn(); + + const selector = new ModelSelectorComponent( + ui, + model, + settings, + modelRegistry, + [{ model, thinkingLevel: "off" }], + () => {}, + onCancel, + ); + + await Bun.sleep(0); + + selector.handleInput("\x1b"); + expect(onCancel).not.toHaveBeenCalled(); + + selector.handleInput("\x07"); + expect(onCancel).toHaveBeenCalledTimes(1); + }); +}); From f09c5bd67c1ef3d29047cffa78cae403db36d7c0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 22:42:42 +0100 Subject: [PATCH 15/22] feat(patch): added boundary duplication detection to prevent off-by-one range errors - Added boundary duplication warning to detect off-by-one range errors in replace_range and replace_line operations. - Updated hashline tool documentation with boundary duplication trap guidance to prevent closing delimiter duplication. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/patch/hashline.ts | 32 +++++++++++++++++++ .../src/prompts/tools/hashline.md | 1 + 3 files changed, 34 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9c9b98b73..1920da9ec 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Renamed hashline edit operation types: `append` → `append_at`, `prepend` → `prepend_at`, `append_eof` → `append_file`, `prepend_bof` → `prepend_file` @@ -11,6 +10,7 @@ ### Added +- Added boundary duplication warning when replace_range or replace_line operations include a last inserted line that matches the next surviving line, helping detect off-by-one range errors - Added git branch isolation for autoresearch sessions via `ensureAutoresearchBranch()` to safely revert failed experiments - Added branch status line to autoresearch initialization and resume prompts showing created or reused branch name - Added `Files in Scope`, `Off Limits`, and `Constraints` sections to autoresearch.md template for explicit scope definition diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index e4c2b6df5..58330de73 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -539,6 +539,38 @@ export function applyHashlineEdits( } maybeAutocorrectEscapedTabIndentation(edits, warnings); maybeWarnSuspiciousUnicodeEscapePlaceholder(edits, warnings); + + // Warn when a replace_range/replace_line's last inserted line duplicates the next surviving line. + // This catches the common boundary-overreach pattern where the agent includes a closing delimiter + // in the replacement but sets `end` to the line before the delimiter, causing duplication. + for (const edit of edits) { + let endLine: number; + switch (edit.op) { + case "replace_line": + endLine = edit.pos.line; + break; + case "replace_range": + endLine = edit.end.line; + break; + default: + continue; + } + if (edit.lines.length === 0) continue; + const nextSurvivingIdx = endLine; // 0-indexed: endLine (1-indexed) is the next line after `end` + if (nextSurvivingIdx >= originalFileLines.length) continue; + const nextSurvivingLine = originalFileLines[nextSurvivingIdx]; + const lastInsertedLine = edit.lines[edit.lines.length - 1]; + const trimmedNext = nextSurvivingLine.trim(); + const trimmedLast = lastInsertedLine.trim(); + // Only warn for non-trivial lines to avoid false positives on blank lines or bare punctuation + if (trimmedLast.length > 0 && trimmedLast === trimmedNext) { + const tag = formatLineTag(endLine + 1, nextSurvivingLine); + warnings.push( + `Possible boundary duplication: your last replacement line \`${trimmedLast}\` is identical to the next surviving line ${tag}. ` + + `If you meant to replace the entire block, set \`end\` to ${tag} instead.`, + ); + } + } // Deduplicate identical edits targeting the same line(s) const seenEditKeys = new Map(); const dedupIndices = new Set(); diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index fcbc3f233..8a5c34d81 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -127,6 +127,7 @@ When adding a sibling declaration, prefer `prepend_at` on the next declaration. - `replace_range` requires both `pos` and `end`. All other anchored ops require `pos` only. - `append_file` and `prepend_file` do not take anchors. - Replace exactly the owned span. If `lines` re-emits content beyond `end`, it will duplicate. +- **Boundary duplication trap**: when replacing a block, `end` must be the **last line of the block** (e.g. the closing `}`), not the last *content* line before it. Otherwise the closing delimiter survives and your replacement adds a second copy. - Do not target shared boundary lines such as `} else {`, `} catch (…) {`, `}),`, or `},{`. - For a block, either replace only the body or replace the whole block. Do not split block boundaries. - `lines` must be literal file content with matching indentation. If the file uses tabs, use real tabs. From 9975f203cdca3683d5996571ceb33c7957c5c63e Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 22 Mar 2026 23:21:01 +0100 Subject: [PATCH 16/22] chore(react-edit-benchmark): increased max-turns to 24 --- packages/react-edit-benchmark/src/index.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/react-edit-benchmark/src/index.ts b/packages/react-edit-benchmark/src/index.ts index e95991add..f41a19b37 100644 --- a/packages/react-edit-benchmark/src/index.ts +++ b/packages/react-edit-benchmark/src/index.ts @@ -67,7 +67,7 @@ Options: --max-attempts Max prompt attempts per run (default: 1) --no-op-retry-limit Stop after repeated preventable no-op failures (default: 2) --mutation-scope-window Allowed line-distance from mutation target for hashline refs (default: 20) - --max-turns Max turn_start events per attempt before failing (default: 10) + --max-turns Max turn_start events per attempt before failing (default: 24) --output Output file (default: run_____.md) --format Output format: markdown, json (default: markdown) --check-fixtures Validate fixtures and exit @@ -152,7 +152,7 @@ async function main(): Promise { thinking: { type: "string", default: "low" }, runs: { type: "string", default: "1" }, timeout: { type: "string", default: "120000" }, - "max-turns": { type: "string", default: "10" }, + "max-turns": { type: "string", default: "24" }, "task-concurrency": { type: "string", default: "16" }, tasks: { type: "string" }, fixtures: { type: "string" }, From 9e824d9235470a33a8e6e090b0dc92f4c5865b25 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 23 Mar 2026 00:51:14 +0100 Subject: [PATCH 17/22] feat(coding-agent): restructured hashline edit schema to nested loc/content objects - Restructured hashline edit schema from flat op/pos/end/lines fields to nested loc/content objects with discriminated union types. - Replaced operation names (replace_line, replace_range, append_at, prepend_at) with unified loc object patterns supporting line, block, append, and prepend anchors. - Updated content field to accept array of strings or null instead of lines parameter in edit entries. - Refactored edit resolution logic to dispatch on loc shape instead of op enum, simplifying location handling. --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/patch/hashline.ts | 2 +- packages/coding-agent/src/patch/index.ts | 131 +++++++++--------- packages/coding-agent/src/patch/shared.ts | 30 ++-- .../src/prompts/tools/hashline.md | 92 +++++------- 5 files changed, 125 insertions(+), 134 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1920da9ec..70e80375f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,8 +1,12 @@ # Changelog ## [Unreleased] + ### Breaking Changes +- Changed hashline edit schema from flat `op`/`pos`/`end`/`lines` fields to structured `loc`/`content` format with location-specific objects +- Renamed hashline edit operations: `replace_line` → `{ line: anchor }`, `replace_range` → `{ block: { pos, end } }`, `append_at` → `{ append: anchor }`, `prepend_at` → `{ prepend: anchor }`, `append_file` → `"append"`, `prepend_file` → `"prepend"` +- Changed `lines` parameter to `content` in hashline edit entries - Renamed hashline edit operation types: `append` → `append_at`, `prepend` → `prepend_at`, `append_eof` → `append_file`, `prepend_bof` → `prepend_file` - Changed hashline edit operation types from `replace` (with optional `end`) to explicit `replace_line` and `replace_range` operations - Added required `append_eof` and `prepend_bof` operations for file-level edits; `append` and `prepend` now require an anchor position diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 58330de73..922eed2b6 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -567,7 +567,7 @@ export function applyHashlineEdits( const tag = formatLineTag(endLine + 1, nextSurvivingLine); warnings.push( `Possible boundary duplication: your last replacement line \`${trimmedLast}\` is identical to the next surviving line ${tag}. ` + - `If you meant to replace the entire block, set \`end\` to ${tag} instead.`, + `If you meant to replace the entire block, set \`end\` to ${tag} instead.`, ); } } diff --git a/packages/coding-agent/src/patch/index.ts b/packages/coding-agent/src/patch/index.ts index 79ee06338..0ccb62b6d 100644 --- a/packages/coding-agent/src/patch/index.ts +++ b/packages/coding-agent/src/patch/index.ts @@ -174,16 +174,35 @@ export function hashlineParseText(edit: string[] | string | null): string[] { return stripNewLinePrefixes(edit); } +const linesSchema = Type.Union([ + Type.Array(Type.String(), { description: "content (preferred format)" }), + Type.String(), + Type.Null(), +]); + +const locSchema = Type.Union( + [ + Type.Literal("append"), + Type.Literal("prepend"), + Type.Object({ append: Type.String({ description: "anchor" }) }), + Type.Object({ prepend: Type.String({ description: "anchor" }) }), + Type.Object({ + line: Type.String({ description: "anchor" }), + }), + Type.Object({ + block: Type.Object({ + pos: Type.String({ description: "anchor" }), + end: Type.String({ description: "limit position" }), + }), + }), + ], + { description: "insert location" }, +); + const hashlineEditSchema = Type.Object( { - op: StringEnum(["replace_line", "replace_range", "append_at", "prepend_at", "append_file", "prepend_file"]), - pos: Type.Optional(Type.String({ description: "anchor" })), - end: Type.Optional(Type.String({ description: "limit position" })), - lines: Type.Union([ - Type.Array(Type.String(), { description: "content (preferred format)" }), - Type.String(), - Type.Null(), - ]), + loc: locSchema, + content: linesSchema, }, { additionalProperties: false }, ); @@ -206,68 +225,48 @@ export type HashlineParams = Static; // ═══════════════════════════════════════════════════════════════════════════ /** - * Map flat tool-schema edits (tag/end) into typed HashlineEdit objects. + * Map loc/content tool-schema edits into typed HashlineEdit objects. * - * Resilient: as long as at least one anchor exists, we execute. - * - replace_line + tag → single-line replace - * - replace_range + tag + end → range replace - * - append_at + tag or end → append after that anchor - * - prepend_at + tag or end → prepend before that anchor - * - append_file → file-level append (no anchors needed) - * - prepend_file → file-level prepend (no anchors needed) - * - * Unknown ops default to replace_line/replace_range based on available anchors. + * Each edit entry has a `loc` (where to edit) and `content` (what to insert/replace). + * loc can be: + * - "append" / "prepend" — file-level insert + * - { append: anchor } / { prepend: anchor } — insert relative to anchor + * - { replace_line: anchor } — replace one line + * - { replace_block: { pos, end } } — replace inclusive range */ function resolveEditAnchors(edits: HashlineToolEdit[]): HashlineEdit[] { const result: HashlineEdit[] = []; for (const edit of edits) { - const lines = hashlineParseText(edit.lines); - const tag = edit.pos ? tryParseTag(edit.pos) : undefined; - const end = edit.end ? tryParseTag(edit.end) : undefined; + const lines = hashlineParseText(edit.content); + const loc = edit.loc; - switch (edit.op) { - case "replace_line": { - const anchor = tag ?? end; - if (!anchor) throw new Error("replace_line requires an anchor (pos)."); - result.push({ op: "replace_line", pos: anchor, lines }); - break; - } - case "replace_range": { - if (!tag || !end) throw new Error("replace_range requires both pos and end anchors."); - result.push({ op: "replace_range", pos: tag, end, lines }); - break; - } - case "append_at": { - const anchor = tag ?? end; - if (!anchor) throw new Error("append_at requires an anchor (pos)."); + if (loc === "append") { + result.push({ op: "append_file", lines }); + } else if (loc === "prepend") { + result.push({ op: "prepend_file", lines }); + } else if (typeof loc === "object") { + if ("append" in loc) { + const anchor = tryParseTag(loc.append); + if (!anchor) throw new Error("append requires a valid anchor."); result.push({ op: "append_at", pos: anchor, lines }); - break; - } - case "prepend_at": { - const anchor = end ?? tag; - if (!anchor) throw new Error("prepend_at requires an anchor (pos)."); + } else if ("prepend" in loc) { + const anchor = tryParseTag(loc.prepend); + if (!anchor) throw new Error("prepend requires a valid anchor."); result.push({ op: "prepend_at", pos: anchor, lines }); - break; - } - case "append_file": { - result.push({ op: "append_file", lines }); - break; - } - case "prepend_file": { - result.push({ op: "prepend_file", lines }); - break; - } - default: { - // Backward compat for stale model output: infer op from available anchors - if (tag && end) { - result.push({ op: "replace_range", pos: tag, end, lines }); - } else if (tag || end) { - result.push({ op: "replace_line", pos: (tag ?? end)!, lines }); - } else { - throw new Error("Unknown op requires at least one anchor (pos or end)."); - } - break; + } else if ("line" in loc) { + const anchor = tryParseTag(loc.line); + if (!anchor) throw new Error("line requires a valid anchor."); + result.push({ op: "replace_line", pos: anchor, lines }); + } else if ("block" in loc) { + const posAnchor = tryParseTag(loc.block.pos); + const endAnchor = tryParseTag(loc.block.end); + if (!posAnchor || !endAnchor) throw new Error("block requires valid pos and end anchors."); + result.push({ op: "replace_range", pos: posAnchor, end: endAnchor, lines }); + } else { + throw new Error("Unknown loc shape. Expected append, prepend, line, or block."); } + } else { + throw new Error(`Invalid loc value: ${JSON.stringify(loc)}`); } } return result; @@ -575,12 +574,10 @@ export class EditTool implements AgentTool { const lines: string[] = []; for (const edit of edits) { // For file creation, only anchorless appends/prepends are valid - if (edit.op === "append_file" || edit.op === "prepend_file") { - if (edit.op === "prepend_file") { - lines.unshift(...hashlineParseText(edit.lines)); - } else { - lines.push(...hashlineParseText(edit.lines)); - } + if (edit.loc === "append") { + lines.push(...hashlineParseText(edit.content)); + } else if (edit.loc === "prepend") { + lines.unshift(...hashlineParseText(edit.content)); } else { throw new Error(`File not found: ${path}`); } diff --git a/packages/coding-agent/src/patch/shared.ts b/packages/coding-agent/src/patch/shared.ts index 6ede7cb2f..8b2ef6361 100644 --- a/packages/coding-agent/src/patch/shared.ts +++ b/packages/coding-agent/src/patch/shared.ts @@ -157,20 +157,28 @@ function formatStreamingHashlineEdits(edits: Partial[], uiThem return { srcLabel: "• (incomplete edit)", dst: "" }; } - const contentLines = Array.isArray(edit.lines) ? (edit.lines as string[]).join("\n") : ""; + const contentLines = Array.isArray(edit.content) ? (edit.content as string[]).join("\n") : ""; + const loc = edit.loc; - const op = typeof edit.op === "string" ? edit.op : "?"; - const pos = typeof edit.pos === "string" ? edit.pos : undefined; - const end = typeof edit.end === "string" ? edit.end : undefined; - - if (pos && end && pos !== end) { - return { srcLabel: `• ${op} ${pos}…${end}`, dst: contentLines }; + if (loc === "append" || loc === "prepend") { + return { srcLabel: `• ${loc} (file-level)`, dst: contentLines }; } - const anchor = pos ?? end; - if (anchor) { - return { srcLabel: `\u2022 ${op} ${anchor}`, dst: contentLines }; + if (typeof loc === "object" && loc) { + if ("block" in loc && typeof loc.block === "object" && loc.block) { + const rb = loc.block as { pos?: string; end?: string }; + return { srcLabel: `• block ${rb.pos ?? "?"}…${rb.end ?? "?"}`, dst: contentLines }; + } + if ("line" in loc) { + return { srcLabel: `• line ${(loc as { line: string }).line}`, dst: contentLines }; + } + if ("append" in loc) { + return { srcLabel: `• append ${(loc as { append: string }).append}`, dst: contentLines }; + } + if ("prepend" in loc) { + return { srcLabel: `• prepend ${(loc as { prepend: string }).prepend}`, dst: contentLines }; + } } - return { srcLabel: `\u2022 ${op} (file-level)`, dst: contentLines }; + return { srcLabel: "• (unknown edit)", dst: contentLines }; } } function formatMetadataLine(lineCount: number | null, language: string | undefined, uiTheme: Theme): string { diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index 8a5c34d81..ff4f3702c 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -2,35 +2,24 @@ Applies precise file edits using `LINE#ID` anchors from `read` output. Read the file first. Copy anchors exactly from the latest `read` output. In one `edit` call, batch all edits for one file. After any successful edit, re-read before editing that file again. -This matters: your output is checked against the real file state. Invalid anchors, invalid op/field combinations, duplicated boundary lines, or semantically equivalent rewrites will fail. +This matters: your output is checked against the real file state. Invalid anchors, duplicated boundary lines, or semantically equivalent rewrites will fail. **Top level** - `path` — file path - `move` — optional rename target - `delete` — optional whole-file delete -- `edits` — array of edit entries +- `edits` — array of `{ loc, content }` entries -**Edit entry shape** -Each entry is: -- `op` — one of `replace_line`, `replace_range`, `append_at`, `prepend_at`, `append_file`, `prepend_file` -- `lines` — replacement/inserted content -- `pos` — required for `replace_line`, `replace_range`, `append_at`, `prepend_at` -- `end` — required only for `replace_range` +**Edit entry**: `{ loc, content }` +- `loc` — where to apply the edit (see below) +- `content` — replacement/inserted lines (array of strings preferred, `null` to delete) -**Meaning** -- `replace_line`: replace exactly one anchored line -- `replace_range`: replace inclusive `pos..end` -- `append_at`: insert after `pos` -- `prepend_at`: insert before `pos` -- `append_file`: insert at end of file -- `prepend_file`: insert at beginning of file - -**`lines`** -- Array of literal file lines is preferred -- `""` means a blank line -- `null` or `[]` deletes for `replace_line` / `replace_range` -- For insert ops, `lines` must contain only the new content +**`loc` values** +- `"append"` / `"prepend"` — insert at end/start of file +- `{ append: "N#ID" }` / `{ prepend: "N#ID" }` — insert after/before anchored line +- `{ line: "N#ID" }` — replace exactly one anchored line +- `{ block: { pos: "N#ID", end: "N#ID" } }` — replace inclusive `pos..end` @@ -56,14 +45,29 @@ All examples below reference the same file, `util.ts`: {{hlinefull 18 "}"}} ``` + +Replace only the catch body. Do not target the shared boundary line `} catch (err) {`. +``` +{ + path: "util.ts", + edits: [{ + loc: { block: { pos: {{hlineref 15 "\t\tconsole.error(err);"}}, end: {{hlineref 16 "\t\treturn null;"}} } }, + content: [ + "\t\tif (isEnoent(err)) return null;", + "\t\tthrow err;" + ] + }] +} +``` + + ``` { path: "util.ts", edits: [{ - op: "replace_line", - pos: {{hlineref 2 "const timeout = 5000;"}}, - lines: ["const timeout = 30_000;"] + loc: { line: {{hlineref 2 "const timeout = 5000;"}} }, + content: ["const timeout = 30_000;"] }] } ``` @@ -74,42 +78,21 @@ All examples below reference the same file, `util.ts`: { path: "util.ts", edits: [{ - op: "replace_range", - pos: {{hlineref 10 "\t// TODO: remove after migration"}}, - end: {{hlineref 11 "\tlegacy();"}}, - lines: null - }] -} -``` - - - -Replace only the catch body. Do not target the shared boundary line `} catch (err) {`. -``` -{ - path: "util.ts", - edits: [{ - op: "replace_range", - pos: {{hlineref 15 "\t\tconsole.error(err);"}}, - end: {{hlineref 16 "\t\treturn null;"}}, - lines: [ - "\t\tif (isEnoent(err)) return null;", - "\t\tthrow err;" - ] + loc: { block: { pos: {{hlineref 10 "\t// TODO: remove after migration"}}, end: {{hlineref 11 "\tlegacy();"}} } }, + content: null }] } ``` -When adding a sibling declaration, prefer `prepend_at` on the next declaration. +When adding a sibling declaration, prefer `prepend` on the next declaration. ``` { path: "util.ts", edits: [{ - op: "prepend_at", - pos: {{hlineref 9 "function beta() {"}}, - lines: [ + loc: { prepend: {{hlineref 9 "function beta() {"}} }, + content: [ "function gamma() {", "\tvalidate();", "}", @@ -124,12 +107,11 @@ When adding a sibling declaration, prefer `prepend_at` on the next declaration. - Make the minimum exact edit. Do not rewrite nearby code unless the consumed range requires it. - Use anchors exactly as `N#ID` from the latest `read` output. -- `replace_range` requires both `pos` and `end`. All other anchored ops require `pos` only. -- `append_file` and `prepend_file` do not take anchors. -- Replace exactly the owned span. If `lines` re-emits content beyond `end`, it will duplicate. +- `block` requires both `pos` and `end`. Other anchored ops require one anchor. +- Replace exactly the owned span. If `content` re-emits content beyond `end`, it will duplicate. - **Boundary duplication trap**: when replacing a block, `end` must be the **last line of the block** (e.g. the closing `}`), not the last *content* line before it. Otherwise the closing delimiter survives and your replacement adds a second copy. - Do not target shared boundary lines such as `} else {`, `} catch (…) {`, `}),`, or `},{`. - For a block, either replace only the body or replace the whole block. Do not split block boundaries. -- `lines` must be literal file content with matching indentation. If the file uses tabs, use real tabs. +- `content` must be literal file content with matching indentation. If the file uses tabs, use real tabs. - Do not use this tool to reformat or clean up unrelated code. - \ No newline at end of file + From 003f46f42ca67ce2e1e66c71625f273de405a775 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 23 Mar 2026 01:49:03 +0100 Subject: [PATCH 18/22] feat(coding-agent/autoresearch): added contract validation and run tracking system - Added contract system for validating benchmark commands, metrics, scope paths, constraints, and off-limits paths. - Contract validation enforces matching initialization parameters against autoresearch.md before init_experiment. - Segment fingerprinting detects configuration drift and warns when metrics are not directly comparable. - Added pending run detection and recovery to resume incomplete experiments from .autoresearch/runs/. - Run directories organize artifacts with benchmark logs and optional checks logs for traceability. - Extended experiment state to track run number, command, scope, off-limits, constraints, and fingerprint. --- packages/coding-agent/CHANGELOG.md | 32 + .../src/autoresearch/command-initialize.md | 18 +- .../src/autoresearch/command-resume.md | 6 + .../coding-agent/src/autoresearch/contract.ts | 318 +++++ .../src/autoresearch/dashboard.ts | 162 ++- packages/coding-agent/src/autoresearch/git.ts | 156 ++- .../coding-agent/src/autoresearch/helpers.ts | 249 +++- .../coding-agent/src/autoresearch/index.ts | 500 ++++++- .../coding-agent/src/autoresearch/prompt.md | 65 +- .../src/autoresearch/resume-message.md | 7 +- .../coding-agent/src/autoresearch/state.ts | 125 +- .../src/autoresearch/tools/init-experiment.ts | 170 ++- .../src/autoresearch/tools/log-experiment.ts | 454 +++++- .../src/autoresearch/tools/run-experiment.ts | 321 +++-- .../coding-agent/src/autoresearch/types.ts | 55 +- .../src/extensibility/extensions/types.ts | 13 +- .../coding-agent/src/session/agent-session.ts | 96 +- .../test/agent-session-concurrent.test.ts | 75 +- .../test/autoresearch-state.test.ts | 665 ++++++++- .../test/autoresearch-tools.test.ts | 1231 +++++++++++++++++ 20 files changed, 4429 insertions(+), 289 deletions(-) create mode 100644 packages/coding-agent/src/autoresearch/contract.ts create mode 100644 packages/coding-agent/test/autoresearch-tools.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 70e80375f..8d165f70b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -14,6 +14,21 @@ ### Added +- Added autoresearch contract system for validating benchmark commands, metrics, scope paths, off-limits paths, and constraints with fingerprint tracking to detect configuration drift +- Added `autoresearch.program.md` support for repo-local playbook overlays that guide session strategy while preserving `autoresearch.md` as source of truth +- Added pending run artifact tracking and recovery to resume incomplete experiments from `.autoresearch/runs/` directory with run numbers and benchmark logs +- Added run directory organization with numbered run artifacts, benchmark logs, and optional checks logs for experiment traceability +- Added segment fingerprinting to detect when benchmark configuration changes between runs and warn about potential incomparability +- Added support for secondary metrics tracking alongside primary metric with configurable direction (lower/higher is better) +- Added `getCurrentAutoresearchBranch()` helper to detect and validate existing autoresearch branches for session resumption +- Added `PendingRunSummary` type to track unlogged run state including parsed metrics, ASI data, and pass/fail status +- Added hidden next-turn message delivery via `deliverAs: 'nextTurn'` with optional `triggerTurn` to queue context for next LLM call without exposing in editable queue +- Added `#queueHiddenNextTurnMessage()` and `#promptQueuedHiddenNextTurnMessages()` to AgentSession for autonomous tool reactions +- Added resume context support in `command-resume.md` template for user-provided guidance when resuming sessions +- Added current segment snapshot display in autoresearch prompt showing recent runs, baseline metrics, and best results +- Added pending run indicator in autoresearch prompt to guide users to complete unlogged experiments before starting new benchmarks +- Added local playbook section in autoresearch prompt when `autoresearch.program.md` exists +- Added tab replacement in dashboard and tool output rendering to prevent display corruption from shell commands with tabs - Added boundary duplication warning when replace_range or replace_line operations include a last inserted line that matches the next surviving line, helping detect off-by-one range errors - Added git branch isolation for autoresearch sessions via `ensureAutoresearchBranch()` to safely revert failed experiments - Added branch status line to autoresearch initialization and resume prompts showing created or reused branch name @@ -47,6 +62,20 @@ ### Changed +- Changed autoresearch initialization to collect and validate benchmark command, metric definition, scope paths, off-limits list, and constraints before `init_experiment` +- Changed `init_experiment` to require exact benchmark command, metric definition, scope, off-limits, and constraints matching collected contract +- Changed `log_experiment` to record run number, benchmark command, scope paths, off-limits list, constraints, and segment fingerprint with each result +- Changed `run_experiment` to organize output in numbered run directories with separate benchmark and checks logs for artifact preservation +- Changed autoresearch dashboard to show pending run indicator when unlogged experiment exists +- Changed autoresearch resume workflow to detect and offer recovery of pending run artifacts before continuing experiment loop +- Changed `ExperimentResult` to include `runNumber`, `benchmarkCommand`, `scopePaths`, `offLimits`, `constraints`, and `segmentFingerprint` fields +- Changed `RunningExperiment` to track `runDirectory` and `runNumber` for artifact organization +- Changed `AutoresearchRuntime` to include `lastRunArtifactDir`, `lastRunNumber`, `lastRunSummary`, `benchmarkCommand`, `secondaryMetrics`, `scopePaths`, `offLimits`, `constraints`, and `segmentFingerprint` +- Changed autoresearch prompts to emphasize `autoresearch.md` as source of truth for benchmark, scope, and constraints +- Changed `command-initialize.md` to display collected setup (benchmark command, metric, direction, scope, off-limits, constraints) before initialization +- Changed `resume-message.md` to reference pending run artifacts and guide completion of unlogged experiments +- Changed `sendMessage()` API documentation to clarify `deliverAs: 'nextTurn'` behavior for hidden context delivery +- Changed `SendMessageHandler` type documentation to explain hidden next-turn message queuing during prompt teardown - Changed autoresearch startup to create or reuse a dedicated `autoresearch/...` git branch before enabling the experiment loop - Changed autoresearch to refuse startup when unrelated worktree changes would make auto-reverts unsafe - Changed autoresearch prompts to emphasize scope and constraints as source of truth for session direction @@ -82,6 +111,9 @@ ### Fixed +- Fixed autoresearch resume to detect and recover pending run artifacts that were left unlogged from previous sessions +- Fixed dashboard overlay to display when running experiment even with zero completed results +- Fixed tab character rendering in dashboard command display and tool output summaries - Fixed autoresearch logging to require durable ASI metadata (hypothesis, rollback_reason, next_action_hint) for every run including rollback context for discarded, crashed, and checks-failed experiments - Fixed autoresearch logging to require durable ASI metadata for every run, including rollback context for discarded, crashed, and checks-failed experiments diff --git a/packages/coding-agent/src/autoresearch/command-initialize.md b/packages/coding-agent/src/autoresearch/command-initialize.md index 271d9135a..9986a844b 100644 --- a/packages/coding-agent/src/autoresearch/command-initialize.md +++ b/packages/coding-agent/src/autoresearch/command-initialize.md @@ -4,13 +4,27 @@ Set up autoresearch for this intent: {{branch_status_line}} +Collected setup: + +- benchmark command: `{{benchmark_command}}` +- primary metric: `{{metric_name}}` +- metric unit: `{{metric_unit}}` +- direction: `{{direction}}` +- files in scope: +{{{scope_paths_block}}} +- off limits: +{{{off_limits_block}}} +- constraints: +{{{constraints_block}}} + Explain briefly what autoresearch will do in this repository, then initialize the workspace. Your first actions: - write `autoresearch.md` -- define `Files in Scope`, `Off Limits`, and `Constraints` in `autoresearch.md` +- record the collected benchmark command, primary metric, metric unit, direction, scope, off-limits list, and constraints in `autoresearch.md` +- optionally write `autoresearch.program.md` when a repo-local playbook would help future resume quality - define the benchmark entrypoint in `autoresearch.sh` - optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate -- run `init_experiment` +- run `init_experiment` with the exact collected benchmark command, metric definition, scope paths, off-limits list, and constraints - run and log the baseline - keep iterating until interrupted or until the configured iteration cap is reached diff --git a/packages/coding-agent/src/autoresearch/command-resume.md b/packages/coding-agent/src/autoresearch/command-resume.md index 46d8cc742..3dd0030a4 100644 --- a/packages/coding-agent/src/autoresearch/command-resume.md +++ b/packages/coding-agent/src/autoresearch/command-resume.md @@ -3,6 +3,12 @@ Resume autoresearch from the attached notes. @{{autoresearch_md_path}} {{branch_status_line}} +{{#if has_resume_context}} + +Additional context from the user: + +{{resume_context}} +{{/if}} Use the notes as the source of truth for the current direction, scope, and constraints. - inspect recent git history for context diff --git a/packages/coding-agent/src/autoresearch/contract.ts b/packages/coding-agent/src/autoresearch/contract.ts new file mode 100644 index 000000000..45d4c4e46 --- /dev/null +++ b/packages/coding-agent/src/autoresearch/contract.ts @@ -0,0 +1,318 @@ +import * as crypto from "node:crypto"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import type { AutoresearchBenchmarkContract, AutoresearchContract, MetricDirection } from "./types"; + +export interface AutoresearchContractLoadResult { + contract: AutoresearchContract; + errors: string[]; + path: string; +} + +export interface AutoresearchScriptSnapshot { + benchmarkScript: string; + benchmarkScriptPath: string; + checksScript: string | null; + checksScriptPath: string; + errors: string[]; +} + +const HEADING_REGEX = /^##\s+(.+?)\s*$/; +const LIST_ITEM_REGEX = /^\s*[-*]\s+(.*)$/; +const KEY_VALUE_REGEX = /^\s*[-*]\s+([^:]+):\s*(.*)$/; + +export function readAutoresearchContract(workDir: string): AutoresearchContractLoadResult { + const contractPath = path.join(workDir, "autoresearch.md"); + let content = ""; + try { + content = fs.readFileSync(contractPath, "utf8"); + } catch { + return { + contract: createEmptyAutoresearchContract(), + errors: [`${contractPath} does not exist. Create it before initializing autoresearch.`], + path: contractPath, + }; + } + + const contract = parseAutoresearchContract(content); + const errors = validateAutoresearchContract(contract); + return { contract, errors, path: contractPath }; +} + +export function parseAutoresearchContract(markdown: string): AutoresearchContract { + const sections = extractSections(markdown); + return { + benchmark: parseBenchmarkSection(sections.get("benchmark") ?? ""), + scopePaths: parseListSection(sections.get("files in scope") ?? "", normalizeContractPathSpec), + offLimits: parseListSection(sections.get("off limits") ?? "", normalizeContractPathSpec), + constraints: parseListSection(sections.get("constraints") ?? ""), + }; +} + +export function validateAutoresearchContract(contract: AutoresearchContract): string[] { + const errors: string[] = []; + if (!contract.benchmark.command) { + errors.push("Benchmark.command is required in autoresearch.md."); + } + if (!contract.benchmark.primaryMetric) { + errors.push("Benchmark.primary metric is required in autoresearch.md."); + } + if (!contract.benchmark.direction) { + errors.push("Benchmark.direction must be `lower` or `higher` in autoresearch.md."); + } + if (contract.scopePaths.length === 0) { + errors.push("Files in Scope must contain at least one path in autoresearch.md."); + } + return errors; +} + +export function buildAutoresearchSegmentFingerprint( + contract: AutoresearchContract, + scripts: { + benchmarkScript: string; + checksScript: string | null; + }, +): string { + const payload = { + benchmark: contract.benchmark, + scopePaths: contract.scopePaths, + offLimits: contract.offLimits, + constraints: contract.constraints, + scripts, + }; + return crypto.createHash("sha256").update(JSON.stringify(payload)).digest("hex"); +} + +export function getAutoresearchFingerprintMismatchError( + stateFingerprint: string | null, + workDir: string, +): string | null { + if (!stateFingerprint) { + return "The current segment has no fingerprint metadata. Re-run init_experiment before continuing."; + } + + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + const errors = [...contractResult.errors, ...scriptSnapshot.errors]; + if (errors.length > 0) { + return `${errors.join(" ")} Re-run init_experiment after fixing the workspace contract.`; + } + + const currentFingerprint = buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); + if (currentFingerprint === stateFingerprint) { + return null; + } + + return "autoresearch.md, autoresearch.sh, or autoresearch.checks.sh changed since the current segment was initialized. Re-run init_experiment before continuing."; +} + +export function loadAutoresearchScriptSnapshot(workDir: string): AutoresearchScriptSnapshot { + const benchmarkScriptPath = path.join(workDir, "autoresearch.sh"); + const checksScriptPath = path.join(workDir, "autoresearch.checks.sh"); + const errors: string[] = []; + + let benchmarkScript = ""; + try { + benchmarkScript = fs.readFileSync(benchmarkScriptPath, "utf8"); + } catch { + errors.push(`${benchmarkScriptPath} does not exist. Create it before initializing autoresearch.`); + } + + let checksScript: string | null = null; + try { + checksScript = fs.readFileSync(checksScriptPath, "utf8"); + } catch { + checksScript = null; + } + + return { + benchmarkScript, + benchmarkScriptPath, + checksScript, + checksScriptPath, + errors, + }; +} + +export function normalizeAutoresearchList(values: readonly string[]): string[] { + const normalized: string[] = []; + const seen = new Set(); + for (const value of values) { + const trimmed = value.trim(); + if (trimmed.length === 0) continue; + if (seen.has(trimmed)) continue; + seen.add(trimmed); + normalized.push(trimmed); + } + return normalized; +} + +export function normalizeContractPathSpec(value: string): string { + const normalized = value.trim().replaceAll("\\", "/"); + if (normalized === "." || normalized === "./") return "."; + return normalized.replace(/^\.\/+/, "").replace(/\/+$/, ""); +} + +export function pathMatchesContractPath(pathValue: string, specValue: string): boolean { + const normalizedPath = normalizeContractPathSpec(pathValue); + const normalizedSpec = normalizeContractPathSpec(specValue); + if (normalizedSpec === ".") return true; + return normalizedPath === normalizedSpec || normalizedPath.startsWith(`${normalizedSpec}/`); +} + +export function contractListsEqual(left: readonly string[], right: readonly string[]): boolean { + const normalizedLeft = normalizeAutoresearchList(left); + const normalizedRight = normalizeAutoresearchList(right); + if (normalizedLeft.length !== normalizedRight.length) return false; + return normalizedLeft.every((value, index) => value === normalizedRight[index]); +} + +export function contractPathListsEqual(left: readonly string[], right: readonly string[]): boolean { + const normalizedLeft = normalizeContractPathList(left); + const normalizedRight = normalizeContractPathList(right); + if (normalizedLeft.length !== normalizedRight.length) return false; + return normalizedLeft.every((value, index) => value === normalizedRight[index]); +} + +function createEmptyAutoresearchContract(): AutoresearchContract { + return { + benchmark: { + command: null, + primaryMetric: null, + metricUnit: "", + direction: null, + secondaryMetrics: [], + }, + scopePaths: [], + offLimits: [], + constraints: [], + }; +} + +function normalizeContractPathList(values: readonly string[]): string[] { + return normalizeAutoresearchList(values.map(normalizeContractPathSpec)).sort((left, right) => + left.localeCompare(right), + ); +} + +function extractSections(markdown: string): Map { + const sections = new Map(); + const lines = markdown.split("\n"); + let currentHeading: string | null = null; + let currentLines: string[] = []; + + for (const line of lines) { + const headingMatch = line.match(HEADING_REGEX); + if (headingMatch) { + if (currentHeading) { + sections.set(currentHeading, currentLines.join("\n").trim()); + } + currentHeading = headingMatch[1]?.trim().toLowerCase() ?? null; + currentLines = []; + continue; + } + if (currentHeading) { + currentLines.push(line); + } + } + + if (currentHeading) { + sections.set(currentHeading, currentLines.join("\n").trim()); + } + return sections; +} + +function parseBenchmarkSection(section: string): AutoresearchBenchmarkContract { + const entries = new Map(); + const lines = section.split("\n"); + for (let index = 0; index < lines.length; index += 1) { + const rawLine = lines[index] ?? ""; + const match = rawLine.match(KEY_VALUE_REGEX); + if (!match) continue; + const key = normalizeKey(match[1] ?? ""); + let value = (match[2] ?? "").trim(); + if (key === "secondarymetrics") { + const nestedItems: string[] = []; + for (let nestedIndex = index + 1; nestedIndex < lines.length; nestedIndex += 1) { + const nestedLine = lines[nestedIndex] ?? ""; + if (nestedLine.match(KEY_VALUE_REGEX)) break; + const nestedMatch = nestedLine.match(/^\s{2,}[-*]\s+(.*)$/); + if (!nestedMatch) { + if (nestedLine.trim().length > 0) break; + continue; + } + nestedItems.push((nestedMatch[1] ?? "").trim()); + index = nestedIndex; + } + if (nestedItems.length > 0) { + value = [value, ...nestedItems].filter(Boolean).join(", "); + } + } + entries.set(key, value); + } + + const direction = parseDirection(entries.get("direction")); + return { + command: readNullableEntry(entries.get("command")), + primaryMetric: readNullableEntry(entries.get("primarymetric")), + metricUnit: entries.get("metricunit")?.trim() ?? "", + direction, + secondaryMetrics: parseSecondaryMetrics(entries.get("secondarymetrics")), + }; +} + +function parseListSection(section: string, normalizeItem?: (value: string) => string): string[] { + const items: string[] = []; + let activeItem: string | null = null; + for (const rawLine of section.split("\n")) { + const line = rawLine.trimEnd(); + if (line.trim().length === 0) continue; + const match = rawLine.match(LIST_ITEM_REGEX); + if (match) { + if (activeItem) items.push(activeItem); + activeItem = (match[1] ?? "").trim(); + continue; + } + if (activeItem && /^\s{2,}\S/.test(rawLine)) { + activeItem = `${activeItem} ${line.trim()}`; + continue; + } + if (activeItem) { + items.push(activeItem); + activeItem = null; + } + items.push(line.trim()); + } + if (activeItem) { + items.push(activeItem); + } + const normalizedItems = normalizeAutoresearchList(items); + return normalizeItem ? normalizedItems.map(normalizeItem) : normalizedItems; +} + +function normalizeKey(value: string): string { + return value.toLowerCase().replace(/[^a-z0-9]+/g, ""); +} + +function parseDirection(value: string | undefined): MetricDirection | null { + if (value === "lower" || value === "higher") return value; + return null; +} + +function readNullableEntry(value: string | undefined): string | null { + const trimmed = value?.trim() ?? ""; + return trimmed.length > 0 ? trimmed : null; +} + +function parseSecondaryMetrics(value: string | undefined): string[] { + if (!value) return []; + return normalizeAutoresearchList( + value + .split(",") + .map(entry => entry.trim()) + .filter(Boolean), + ); +} diff --git a/packages/coding-agent/src/autoresearch/dashboard.ts b/packages/coding-agent/src/autoresearch/dashboard.ts index 246277baf..2f46cb268 100644 --- a/packages/coding-agent/src/autoresearch/dashboard.ts +++ b/packages/coding-agent/src/autoresearch/dashboard.ts @@ -1,5 +1,6 @@ import { matchesKey, Text, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui"; import type { Theme } from "../modes/theme/theme"; +import { replaceTabs } from "../tools/render-utils"; import { formatElapsed, formatNum, isBetter } from "./helpers"; import { currentResults, findBaselineMetric, findBaselineRunNumber, findBaselineSecondary } from "./state"; import type { AutoresearchRuntime, DashboardController, ExperimentResult, ExperimentState } from "./types"; @@ -32,7 +33,7 @@ export function createDashboardController(): DashboardController { updateWidget(ctx, runtime): void { if (!ctx.hasUI) return; const state = runtime.state; - if (state.results.length === 0 && !runtime.runningExperiment) { + if (!shouldShowDashboard(runtime, state)) { ctx.ui.setWidget("autoresearch", undefined); return; } @@ -44,8 +45,8 @@ export function createDashboardController(): DashboardController { if (runtime.dashboardExpanded) { const width = process.stdout.columns ?? 120; const lines = [ - renderExpandedHeader(state, width, theme), - ...renderDashboardLines(state, width, theme, 8), + renderExpandedHeader(runtime, width, theme), + ...renderDashboardLines(runtime, width, theme, 8), ]; return new Text(lines.join("\n"), 0, 0); } @@ -53,7 +54,7 @@ export function createDashboardController(): DashboardController { }); }, async showOverlay(ctx, runtime): Promise { - if (!ctx.hasUI || runtime.state.results.length === 0) return; + if (!ctx.hasUI || !shouldShowDashboard(runtime, runtime.state)) return; await ctx.ui.custom( (tui, theme, _keybindings, done) => { overlayTui = tui; @@ -68,8 +69,8 @@ export function createDashboardController(): DashboardController { return { render(width: number): string[] { const terminalRows = process.stdout.rows ?? 40; - const header = renderExpandedHeader(runtime.state, width, theme); - const body = renderDashboardLines(runtime.state, width, theme, 0); + const header = renderExpandedHeader(runtime, width, theme); + const body = renderDashboardLines(runtime, width, theme, 0); if (runtime.runningExperiment) { body.push(renderOverlayRunningLine(runtime, theme, width, spinnerFrame)); } @@ -87,7 +88,7 @@ export function createDashboardController(): DashboardController { }, handleInput(data: string): void { const totalRows = - renderDashboardLines(runtime.state, process.stdout.columns ?? 120, theme, 0).length + + renderDashboardLines(runtime, process.stdout.columns ?? 120, theme, 0).length + (runtime.runningExperiment ? 1 : 0); const viewportRows = Math.max(4, (process.stdout.rows ?? 40) - 4); const maxScroll = Math.max(0, totalRows - viewportRows); @@ -125,41 +126,87 @@ export function createDashboardController(): DashboardController { function renderRunningOnly(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { const parts = [theme.fg("accent", "autoresearch"), theme.fg("warning", " running...")]; if (state.name) { - parts.push(theme.fg("dim", ` | ${state.name}`)); + parts.push(theme.fg("dim", ` | ${replaceTabs(state.name)}`)); } if (runtime.runningExperiment) { - parts.push(theme.fg("dim", ` | ${runtime.runningExperiment.command}`)); + parts.push(theme.fg("dim", ` | ${replaceTabs(runtime.runningExperiment.command)}`)); } return parts.join(""); } -function renderExpandedHeader(state: ExperimentState, width: number, theme: Theme): string { - const label = state.name ? ` autoresearch: ${state.name} ` : " autoresearch "; - const hint = theme.fg("dim", " ctrl+x collapse ctrl+shift+x fullscreen "); +function shouldShowDashboard(runtime: AutoresearchRuntime, state: ExperimentState): boolean { + return ( + runtime.autoresearchMode || + state.results.length > 0 || + runtime.runningExperiment !== null || + runtime.lastRunSummary !== null + ); +} + +function renderExpandedHeader(runtime: AutoresearchRuntime, width: number, theme: Theme): string { + const state = runtime.state; + const status = renderModeStatus(runtime, state); + const label = state.name ? ` autoresearch: ${replaceTabs(state.name)} ` : " autoresearch "; + const hint = theme.fg("dim", ` ctrl+x collapse ctrl+shift+x overlay${status ? ` ${status}` : ""} `); const fillWidth = Math.max(0, width - visibleWidth(label) - visibleWidth(hint)); return truncateToWidth(theme.fg("accent", label) + theme.fg("borderMuted", "-".repeat(fillWidth)) + hint, width); } function renderCollapsedLine(runtime: AutoresearchRuntime, state: ExperimentState, theme: Theme): string { + if (runtime.lastRunSummary) { + const parts = [ + theme.fg("accent", "autoresearch"), + theme.fg("warning", ` pending run #${runtime.lastRunSummary.runNumber}`), + theme.fg("dim", runtime.lastRunSummary.passed ? " pass" : " fail"), + ]; + if (runtime.lastRunSummary.parsedPrimary !== null) { + parts.push( + theme.fg( + "muted", + ` | ${state.metricName}=${formatNum(runtime.lastRunSummary.parsedPrimary, state.metricUnit)}`, + ), + ); + } + parts.push(theme.fg("warning", " | log_experiment required")); + if (!runtime.autoresearchMode) { + parts.push(theme.fg("dim", " | mode off")); + } + return parts.join(""); + } + if (state.results.length === 0) { + const modeStatus = runtime.autoresearchMode ? "baseline pending" : "mode off"; + const parts = [theme.fg("accent", "autoresearch"), theme.fg("warning", ` ${modeStatus}`)]; + if (state.name) { + parts.push(theme.fg("dim", ` | ${replaceTabs(state.name)}`)); + } + if (runtime.autoresearchMode) { + parts.push(theme.fg("dim", " | run the baseline")); + } + return parts.join(""); + } const current = currentResults(state.results, state.currentSegment); const kept = current.filter(result => result.status === "keep").length; const crashed = current.filter(result => result.status === "crash").length; const checksFailed = current.filter(result => result.status === "checks_failed").length; const best = findBestResult(state); + const archivedRuns = Math.max(0, state.results.length - current.length); const parts = [ theme.fg("accent", "autoresearch"), - theme.fg("muted", ` ${state.results.length} runs`), + theme.fg("muted", ` ${current.length} runs`), theme.fg("success", ` ${kept} kept`), ]; + if (archivedRuns > 0) parts.push(theme.fg("dim", ` +${archivedRuns} archived`)); if (crashed > 0) parts.push(theme.fg("error", ` ${crashed} crash`)); if (checksFailed > 0) parts.push(theme.fg("error", ` ${checksFailed} checks_failed`)); parts.push(theme.fg("dim", " | ")); - parts.push( - theme.fg( - "warning", - `${state.metricName}: ${formatNum(best?.result.metric ?? state.bestMetric, state.metricUnit)}`, - ), - ); + if (best && state.bestMetric !== null && best.result.metric !== state.bestMetric) { + parts.push(theme.fg("warning", `best ${formatNum(best.result.metric, state.metricUnit)}`)); + parts.push(theme.fg("dim", ` baseline ${formatNum(state.bestMetric, state.metricUnit)}`)); + } else if (state.bestMetric !== null) { + parts.push(theme.fg("warning", `baseline ${formatNum(state.bestMetric, state.metricUnit)}`)); + } else { + parts.push(theme.fg("warning", `no kept runs yet`)); + } if (state.confidence !== null) { const confidenceColor = state.confidence >= 2 ? "success" : state.confidence >= 1 ? "warning" : "error"; parts.push(theme.fg("dim", " | ")); @@ -167,13 +214,42 @@ function renderCollapsedLine(runtime: AutoresearchRuntime, state: ExperimentStat } if (runtime.runningExperiment) { parts.push(theme.fg("dim", ` | running ${formatElapsed(Date.now() - runtime.runningExperiment.startedAt)}`)); + } else if (!runtime.autoresearchMode) { + parts.push(theme.fg("dim", ` | ${renderModeStatus(runtime, state)}`)); } parts.push(theme.fg("dim", " | ctrl+x expand")); return parts.join(""); } -export function renderDashboardLines(state: ExperimentState, width: number, theme: Theme, maxRows: number): string[] { +export function renderDashboardLines( + runtime: AutoresearchRuntime, + width: number, + theme: Theme, + maxRows: number, +): string[] { + const state = runtime.state; if (state.results.length === 0) { + if (runtime.lastRunSummary) { + const lines = [ + truncateToWidth(`Pending run: #${runtime.lastRunSummary.runNumber}`, width), + truncateToWidth( + `Result: ${runtime.lastRunSummary.passed ? "passed" : "failed"}${runtime.lastRunSummary.parsedPrimary !== null ? ` ${state.metricName} ${formatNum(runtime.lastRunSummary.parsedPrimary, state.metricUnit)}` : ""}`, + width, + ), + truncateToWidth("Next action: finish log_experiment before starting another run.", width), + ]; + if (!runtime.autoresearchMode) { + lines.push(truncateToWidth("Mode: off", width)); + } + return lines; + } + if (runtime.autoresearchMode) { + return [ + truncateToWidth("Current segment: 0 runs", width), + truncateToWidth("Baseline: pending", width), + truncateToWidth("Next action: run and log the baseline experiment.", width), + ]; + } return [theme.fg("dim", "No experiments logged yet.")]; } @@ -188,7 +264,7 @@ export function renderDashboardLines(state: ExperimentState, width: number, them const best = findBestResult(state); const lines = [ truncateToWidth( - `Runs: ${state.results.length} ${kept} kept ${discarded} discarded ${crashed} crashed ${checksFailed} checks_failed`, + `Current segment: ${current.length} runs ${kept} kept ${discarded} discarded ${crashed} crashed ${checksFailed} checks_failed`, width, ), truncateToWidth( @@ -196,8 +272,25 @@ export function renderDashboardLines(state: ExperimentState, width: number, them width, ), ]; + if (state.results.length > current.length) { + lines.push( + truncateToWidth(`Archived from earlier segments: ${state.results.length - current.length} runs`, width), + ); + } + if (runtime.lastRunSummary) { + lines.push( + truncateToWidth( + `Pending run: #${runtime.lastRunSummary.runNumber} (${runtime.lastRunSummary.passed ? "passed" : "failed"}) — log_experiment required`, + width, + ), + ); + } + if (!runtime.autoresearchMode) { + lines.push(truncateToWidth(`Mode: ${renderModeStatus(runtime, state)}`, width)); + } if (best) { - let progress = `Best: ${formatNum(best.result.metric, state.metricUnit)} (#${best.index + 1})`; + const bestRunNumber = best.result.runNumber ?? best.index + 1; + let progress = `Best: ${formatNum(best.result.metric, state.metricUnit)} (#${bestRunNumber})`; if (baseline !== null && baseline !== 0 && best.result.metric !== baseline) { const delta = ((best.result.metric - baseline) / baseline) * 100; const sign = delta > 0 ? "+" : ""; @@ -227,9 +320,9 @@ export function renderDashboardLines(state: ExperimentState, width: number, them lines.push(renderTableHeader(state, width, theme)); lines.push(theme.fg("borderMuted", "-".repeat(Math.max(0, width - 1)))); - const visible = maxRows > 0 ? state.results.slice(-maxRows) : state.results; - if (visible.length < state.results.length) { - lines.push(theme.fg("dim", `... ${state.results.length - visible.length} earlier runs hidden ...`)); + const visible = maxRows > 0 ? current.slice(-maxRows) : current; + if (visible.length < current.length) { + lines.push(theme.fg("dim", `... ${current.length - visible.length} earlier runs hidden ...`)); } for (const result of visible) { lines.push(renderResultRow(result, state, baselineSecondary, width, theme)); @@ -252,7 +345,7 @@ function renderResultRow( width: number, theme: Theme, ): string { - const runNumber = state.results.indexOf(result) + 1; + const runNumber = result.runNumber ?? state.results.indexOf(result) + 1; const secondary = state.secondaryMetrics .map(metric => truncateToWidth( @@ -268,7 +361,7 @@ function renderResultRow( `${theme.fg(statusColor, formatNum(result.metric, state.metricUnit).padEnd(12))}` + `${secondary}` + `${theme.fg(statusColor, result.status.padEnd(14))}` + - `${theme.fg("muted", result.description)}`; + `${theme.fg("muted", replaceTabs(result.description))}`; return truncateToWidth(line, width); } @@ -306,7 +399,9 @@ function renderOverlayRunningLine( return truncateToWidth( theme.fg( "warning", - `${spinner} running ${formatElapsed(Date.now() - (runtime.runningExperiment?.startedAt ?? Date.now()))} ${runtime.runningExperiment?.command ?? ""}`, + `${spinner} running ${formatElapsed(Date.now() - (runtime.runningExperiment?.startedAt ?? Date.now()))} ${replaceTabs( + runtime.runningExperiment?.command ?? "", + )}`, ), width, ); @@ -328,6 +423,17 @@ function renderOverlayFooter( return theme.fg("borderMuted", "-".repeat(fill)) + hint; } +function renderModeStatus(runtime: AutoresearchRuntime, state: ExperimentState): string { + if (runtime.autoresearchMode) { + return state.results.length === 0 ? "baseline pending" : "mode on"; + } + const current = currentResults(state.results, state.currentSegment); + if (state.maxExperiments !== null && current.length >= state.maxExperiments) { + return "segment complete"; + } + return "mode off"; +} + function findBestResult(state: ExperimentState): { index: number; result: ExperimentResult } | null { let best: { index: number; result: ExperimentResult } | null = null; for (let index = 0; index < state.results.length; index += 1) { diff --git a/packages/coding-agent/src/autoresearch/git.ts b/packages/coding-agent/src/autoresearch/git.ts index 12caf3721..e22ea4976 100644 --- a/packages/coding-agent/src/autoresearch/git.ts +++ b/packages/coding-agent/src/autoresearch/git.ts @@ -1,5 +1,5 @@ import type { ExtensionAPI } from "../extensibility/extensions"; -import { PROTECTED_AUTORESEARCH_FILES } from "./helpers"; +import { isAutoresearchLocalStatePath, normalizeAutoresearchPath } from "./helpers"; const AUTORESEARCH_BRANCH_PREFIX = "autoresearch/"; const BRANCH_NAME_MAX_LENGTH = 48; @@ -17,6 +17,12 @@ export interface EnsureAutoresearchBranchSuccess { export type EnsureAutoresearchBranchResult = EnsureAutoresearchBranchFailure | EnsureAutoresearchBranchSuccess; +export async function getCurrentAutoresearchBranch(api: ExtensionAPI, workDir: string): Promise { + const currentBranchResult = await api.exec("git", ["branch", "--show-current"], { cwd: workDir, timeout: 5_000 }); + const currentBranch = currentBranchResult.stdout.trim(); + return currentBranch.startsWith(AUTORESEARCH_BRANCH_PREFIX) ? currentBranch : null; +} + export async function ensureAutoresearchBranch( api: ExtensionAPI, workDir: string, @@ -29,19 +35,10 @@ export async function ensureAutoresearchBranch( ok: false, }; } + const repoRoot = repoRootResult.stdout.trim() || workDir; - const currentBranchResult = await api.exec("git", ["branch", "--show-current"], { cwd: workDir, timeout: 5_000 }); - const currentBranch = currentBranchResult.stdout.trim(); - if (currentBranch.startsWith(AUTORESEARCH_BRANCH_PREFIX)) { - return { - branchName: currentBranch, - created: false, - ok: true, - }; - } - - const dirtyPathsResult = await api.exec("git", ["status", "--porcelain", "--untracked-files=all"], { - cwd: workDir, + const dirtyPathsResult = await api.exec("git", ["status", "--porcelain=v1", "-z", "--untracked-files=all"], { + cwd: repoRoot, timeout: 5_000, }); if (dirtyPathsResult.code !== 0) { @@ -51,17 +48,22 @@ export async function ensureAutoresearchBranch( }; } - const unsafeDirtyPaths = parseUnsafeDirtyPaths(dirtyPathsResult.stdout); - if (unsafeDirtyPaths.length > 0) { - const preview = unsafeDirtyPaths.slice(0, 5).join(", "); - const suffix = unsafeDirtyPaths.length > 5 ? ` (+${unsafeDirtyPaths.length - 5} more)` : ""; + const workDirPrefix = await readGitWorkDirPrefix(api, workDir); + const unsafeDirtyPaths = collectUnsafeDirtyPaths(dirtyPathsResult.stdout, workDirPrefix); + const currentBranch = await getCurrentAutoresearchBranch(api, workDir); + if (currentBranch) { + if (unsafeDirtyPaths.length > 0) { + return buildUnsafeDirtyPathsFailure(unsafeDirtyPaths); + } return { - error: - "Autoresearch needs a clean git worktree before it can create an isolated branch. " + - `Commit or stash these paths first: ${preview}${suffix}`, - ok: false, + branchName: currentBranch, + created: false, + ok: true, }; } + if (unsafeDirtyPaths.length > 0) { + return buildUnsafeDirtyPathsFailure(unsafeDirtyPaths); + } const branchName = await allocateBranchName(api, workDir, goal); const checkoutResult = await api.exec("git", ["checkout", "-b", branchName], { cwd: workDir, timeout: 10_000 }); @@ -81,7 +83,69 @@ export async function ensureAutoresearchBranch( }; } -function parseUnsafeDirtyPaths(statusOutput: string): string[] { +export function parseWorkDirDirtyPaths(statusOutput: string, workDirPrefix: string): string[] { + const relativePaths: string[] = []; + for (const dirtyPath of parseDirtyPaths(statusOutput)) { + const relativePath = relativizeGitPathToWorkDir(dirtyPath, workDirPrefix); + if (relativePath === null) continue; + relativePaths.push(relativePath); + } + return relativePaths; +} + +export function relativizeGitPathToWorkDir(repoRelativePath: string, workDirPrefix: string): string | null { + const normalizedPath = normalizeStatusPath(repoRelativePath); + const normalizedPrefix = normalizeAutoresearchPath(workDirPrefix); + if (normalizedPrefix === "" || normalizedPrefix === ".") { + return normalizedPath; + } + if (normalizedPath === normalizedPrefix) { + return "."; + } + if (!normalizedPath.startsWith(`${normalizedPrefix}/`)) { + return null; + } + return normalizeAutoresearchPath(normalizedPath.slice(normalizedPrefix.length + 1)); +} + +async function readGitWorkDirPrefix(api: ExtensionAPI, workDir: string): Promise { + const prefixResult = await api.exec("git", ["rev-parse", "--show-prefix"], { cwd: workDir, timeout: 5_000 }); + if (prefixResult.code !== 0) { + return ""; + } + return prefixResult.stdout.trim(); +} + +export function parseDirtyPaths(statusOutput: string): string[] { + if (statusOutput.includes("\0")) { + return parseDirtyPathsNul(statusOutput); + } + return parseDirtyPathsLines(statusOutput); +} + +function parseDirtyPathsNul(statusOutput: string): string[] { + const unsafePaths = new Set(); + let index = 0; + while (index + 3 <= statusOutput.length) { + const statusToken = statusOutput.slice(index, index + 3); + index += 3; + const pathEnd = statusOutput.indexOf("\0", index); + if (pathEnd < 0) break; + const firstPath = statusOutput.slice(index, pathEnd); + index = pathEnd + 1; + addDirtyPath(unsafePaths, firstPath); + if (isRenameOrCopy(statusToken)) { + const secondPathEnd = statusOutput.indexOf("\0", index); + if (secondPathEnd < 0) break; + const secondPath = statusOutput.slice(index, secondPathEnd); + index = secondPathEnd + 1; + addDirtyPath(unsafePaths, secondPath); + } + } + return [...unsafePaths]; +} + +function parseDirtyPathsLines(statusOutput: string): string[] { const unsafePaths = new Set(); for (const line of statusOutput.split("\n")) { const trimmedLine = line.trimEnd(); @@ -89,23 +153,19 @@ function parseUnsafeDirtyPaths(statusOutput: string): string[] { const rawPath = trimmedLine.slice(3).trim(); if (rawPath.length === 0) continue; const renameParts = rawPath.split(" -> "); - const normalizedPath = normalizeStatusPath(renameParts[renameParts.length - 1] ?? rawPath); - if (normalizedPath.length === 0) continue; - if (PROTECTED_AUTORESEARCH_FILES.some(path => path === normalizedPath)) continue; - unsafePaths.add(normalizedPath); + for (const renamePart of renameParts) { + addDirtyPath(unsafePaths, renamePart); + } } return [...unsafePaths]; } -function normalizeStatusPath(path: string): string { +export function normalizeStatusPath(path: string): string { let normalized = path.trim(); if (normalized.startsWith('"') && normalized.endsWith('"')) { normalized = normalized.slice(1, -1); } - if (normalized.startsWith("./")) { - normalized = normalized.slice(2); - } - return normalized; + return normalizeAutoresearchPath(normalized); } async function allocateBranchName(api: ExtensionAPI, workDir: string, goal: string | null): Promise { @@ -147,3 +207,37 @@ function currentDateStamp(): string { function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { return `${result.stdout}${result.stderr}`; } + +function addDirtyPath(paths: Set, rawPath: string): void { + const normalizedPath = normalizeStatusPath(rawPath); + if (normalizedPath.length === 0) return; + paths.add(normalizedPath); +} + +function buildUnsafeDirtyPathsFailure(unsafeDirtyPaths: string[]): EnsureAutoresearchBranchFailure { + const preview = unsafeDirtyPaths.slice(0, 5).join(", "); + const suffix = unsafeDirtyPaths.length > 5 ? ` (+${unsafeDirtyPaths.length - 5} more)` : ""; + return { + error: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. " + + `Commit or stash these paths first: ${preview}${suffix}`, + ok: false, + }; +} + +function isRenameOrCopy(statusToken: string): boolean { + const trimmed = statusToken.trim(); + return trimmed.startsWith("R") || trimmed.startsWith("C"); +} + +function collectUnsafeDirtyPaths(statusOutput: string, workDirPrefix: string): string[] { + const unsafeDirtyPaths: string[] = []; + for (const dirtyPath of parseDirtyPaths(statusOutput)) { + const relativePath = relativizeGitPathToWorkDir(dirtyPath, workDirPrefix); + if (relativePath && isAutoresearchLocalStatePath(relativePath)) { + continue; + } + unsafeDirtyPaths.push(relativePath ?? normalizeStatusPath(dirtyPath)); + } + return unsafeDirtyPaths; +} diff --git a/packages/coding-agent/src/autoresearch/helpers.ts b/packages/coding-agent/src/autoresearch/helpers.ts index 681c50f41..20f2ae0f7 100644 --- a/packages/coding-agent/src/autoresearch/helpers.ts +++ b/packages/coding-agent/src/autoresearch/helpers.ts @@ -1,21 +1,28 @@ -import * as crypto from "node:crypto"; import * as fs from "node:fs"; -import * as os from "node:os"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; -import type { ASIData, ASIValue, AutoresearchConfig, MetricDirection } from "./types"; +import type { + ASIData, + ASIValue, + AutoresearchConfig, + MetricDirection, + NumericMetricMap, + PendingRunSummary, +} from "./types"; export const METRIC_LINE_PREFIX = "METRIC"; export const ASI_LINE_PREFIX = "ASI"; export const EXPERIMENT_MAX_LINES = 10; export const EXPERIMENT_MAX_BYTES = 4 * 1024; -export const PROTECTED_AUTORESEARCH_FILES = [ - "autoresearch.jsonl", +export const AUTORESEARCH_COMMITTABLE_FILES = [ "autoresearch.md", - "autoresearch.ideas.md", + "autoresearch.program.md", "autoresearch.sh", "autoresearch.checks.sh", + "autoresearch.ideas.md", ] as const; +export const AUTORESEARCH_LOCAL_STATE_FILES = ["autoresearch.jsonl"] as const; +export const AUTORESEARCH_LOCAL_STATE_DIRECTORIES = [".autoresearch"] as const; const DENIED_KEY_NAMES = new Set(["__proto__", "constructor", "prototype"]); @@ -112,21 +119,57 @@ export function formatElapsed(milliseconds: number): string { return `${seconds}s`; } -export function createTempFileAllocator(): () => string { - let tempPath: string | undefined; - return () => { - if (tempPath) return tempPath; - tempPath = path.join(os.tmpdir(), `pi-autoresearch-${crypto.randomUUID()}.log`); - return tempPath; - }; +export function getAutoresearchRunDirectory(workDir: string, runNumber: number): string { + return path.join(workDir, ".autoresearch", "runs", String(runNumber).padStart(4, "0")); } -export function killTree(pid: number): void { +export function getNextAutoresearchRunNumber(workDir: string, lastRunNumber: number | null): number { + const runsDirectory = path.join(workDir, ".autoresearch", "runs"); + let maxRunNumber = lastRunNumber ?? 0; try { - process.kill(-pid, "SIGTERM"); + for (const entry of fs.readdirSync(runsDirectory, { withFileTypes: true })) { + if (!entry.isDirectory()) continue; + const runNumber = Number.parseInt(entry.name, 10); + if (Number.isFinite(runNumber)) { + maxRunNumber = Math.max(maxRunNumber, runNumber); + } + } + } catch (error) { + if (!isEnoent(error)) { + throw error; + } + } + return maxRunNumber + 1; +} + +export function normalizeAutoresearchPath(relativePath: string): string { + const normalized = relativePath.replaceAll("\\", "/").trim(); + if (normalized === "." || normalized === "./") return "."; + return normalized.replace(/^\.\/+/, "").replace(/\/+$/, ""); +} + +export function isAutoresearchCommittableFile(relativePath: string): boolean { + const normalized = normalizeAutoresearchPath(relativePath); + return AUTORESEARCH_COMMITTABLE_FILES.some(candidate => candidate === normalized); +} + +export function isAutoresearchLocalStatePath(relativePath: string): boolean { + const normalized = normalizeAutoresearchPath(relativePath); + if (AUTORESEARCH_LOCAL_STATE_FILES.some(candidate => candidate === normalized)) { + return true; + } + return AUTORESEARCH_LOCAL_STATE_DIRECTORIES.some(candidate => { + const normalizedCandidate = normalizeAutoresearchPath(candidate); + return normalized === normalizedCandidate || normalized.startsWith(`${normalizedCandidate}/`); + }); +} + +export function killTree(pid: number, signal: NodeJS.Signals | number = "SIGTERM"): void { + try { + process.kill(-pid, signal); } catch { try { - process.kill(pid, "SIGTERM"); + process.kill(pid, signal); } catch { // Process already exited. } @@ -159,6 +202,44 @@ export function inferMetricUnitFromName(name: string): string { return ""; } +export async function readPendingRunSummary( + workDir: string, + loggedRunNumbers: ReadonlySet = new Set(), +): Promise { + const runsDir = path.join(workDir, ".autoresearch", "runs"); + let entries: fs.Dirent[]; + try { + entries = await fs.promises.readdir(runsDir, { withFileTypes: true }); + } catch (error) { + if (isEnoent(error)) return null; + throw error; + } + + const runDirectories = entries + .filter(entry => entry.isDirectory()) + .map(entry => entry.name) + .sort((left, right) => right.localeCompare(left)); + + for (const directoryName of runDirectories) { + const runDirectory = path.join(runsDir, directoryName); + const runJsonPath = path.join(runDirectory, "run.json"); + let parsed: unknown; + try { + parsed = await Bun.file(runJsonPath).json(); + } catch (error) { + if (isEnoent(error)) continue; + throw error; + } + + const pendingRun = parsePendingRunSummary(parsed, runDirectory, directoryName, loggedRunNumbers); + if (pendingRun) { + return pendingRun; + } + } + + return null; +} + export function readConfig(cwd: string): AutoresearchConfig { const configPath = path.join(cwd, "autoresearch.config.json"); try { @@ -207,3 +288,139 @@ export function validateWorkDir(cwd: string): string | null { return `workingDir ${workDir} is unavailable.`; } } + +function parsePendingRunSummary( + value: unknown, + runDirectory: string, + directoryName: string, + loggedRunNumbers: ReadonlySet, +): PendingRunSummary | null { + if (typeof value !== "object" || value === null) return null; + const candidate = value as { + checks?: { durationSeconds?: unknown; passed?: unknown; timedOut?: unknown }; + completedAt?: unknown; + command?: unknown; + durationSeconds?: unknown; + exitCode?: unknown; + loggedAt?: unknown; + parsedAsi?: unknown; + parsedMetrics?: unknown; + parsedPrimary?: unknown; + runNumber?: unknown; + status?: unknown; + timedOut?: unknown; + }; + if (candidate.loggedAt !== undefined || candidate.status !== undefined) { + return null; + } + + const command = typeof candidate.command === "string" ? candidate.command : ""; + const runNumber = + typeof candidate.runNumber === "number" && Number.isFinite(candidate.runNumber) + ? candidate.runNumber + : parseInt(directoryName, 10); + if (!Number.isFinite(runNumber)) return null; + if (loggedRunNumbers.has(runNumber)) return null; + + const hasCompletedMetadata = + typeof candidate.completedAt === "string" || + candidate.exitCode !== undefined || + candidate.timedOut !== undefined || + candidate.durationSeconds !== undefined || + candidate.checks !== undefined || + candidate.parsedPrimary !== undefined || + candidate.parsedMetrics !== undefined || + candidate.parsedAsi !== undefined; + if (!hasCompletedMetadata) { + return null; + } + + const checksPass = + typeof candidate.checks?.passed === "boolean" + ? candidate.checks.passed + : typeof candidate.checks?.timedOut === "boolean" && candidate.checks.timedOut + ? false + : null; + const exitCode = + typeof candidate.exitCode === "number" && Number.isFinite(candidate.exitCode) ? candidate.exitCode : null; + const timedOut = candidate.timedOut === true; + const durationSeconds = + typeof candidate.durationSeconds === "number" && Number.isFinite(candidate.durationSeconds) + ? candidate.durationSeconds + : null; + const parsedPrimary = + typeof candidate.parsedPrimary === "number" && Number.isFinite(candidate.parsedPrimary) + ? candidate.parsedPrimary + : null; + const parsedAsi = cloneAsiData(candidate.parsedAsi); + const parsedMetrics = cloneNumericMetricMap(candidate.parsedMetrics); + const checksDurationSeconds = + typeof candidate.checks?.durationSeconds === "number" && Number.isFinite(candidate.checks.durationSeconds) + ? candidate.checks.durationSeconds + : null; + const checksTimedOut = candidate.checks?.timedOut === true; + + return { + checksDurationSeconds, + checksPass, + checksTimedOut, + command, + durationSeconds, + parsedAsi, + parsedMetrics, + parsedPrimary, + passed: exitCode === 0 && !timedOut && checksPass !== false, + runDirectory, + runNumber, + }; +} + +function cloneNumericMetricMap(value: unknown): NumericMetricMap | null { + if (typeof value !== "object" || value === null) return null; + const metrics = value as { [key: string]: unknown }; + const clone: NumericMetricMap = {}; + for (const [key, entryValue] of Object.entries(metrics)) { + if (typeof entryValue === "number" && Number.isFinite(entryValue)) { + clone[key] = entryValue; + } + } + return Object.keys(clone).length > 0 ? clone : null; +} + +function cloneAsiData(value: unknown): ASIData | null { + if (typeof value !== "object" || value === null) return null; + const candidate = value as { [key: string]: unknown }; + const clone: ASIData = {}; + for (const [key, entryValue] of Object.entries(candidate)) { + const sanitized = clonePendingAsiValue(entryValue); + if (sanitized !== undefined) { + clone[key] = sanitized; + } + } + return Object.keys(clone).length > 0 ? clone : null; +} + +function clonePendingAsiValue(value: unknown): ASIValue | undefined { + if (value === null) return null; + if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") { + return value; + } + if (Array.isArray(value)) { + const items = value + .map(entry => clonePendingAsiValue(entry)) + .filter((entry): entry is NonNullable => entry !== undefined); + return items; + } + if (typeof value === "object") { + const candidate = value as { [key: string]: unknown }; + const clone: { [key: string]: ASIValue } = {}; + for (const [key, entryValue] of Object.entries(candidate)) { + const sanitized = clonePendingAsiValue(entryValue); + if (sanitized !== undefined) { + clone[key] = sanitized; + } + } + return clone; + } + return undefined; +} diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 6de676653..4ddcbd2c2 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -5,27 +5,49 @@ import { renderPromptTemplate } from "../config/prompt-templates"; import type { ExtensionContext, ExtensionFactory } from "../extensibility/extensions"; import commandInitializeTemplate from "./command-initialize.md" with { type: "text" }; import commandResumeTemplate from "./command-resume.md" with { type: "text" }; +import { pathMatchesContractPath } from "./contract"; import { createDashboardController } from "./dashboard"; import { ensureAutoresearchBranch } from "./git"; -import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "./helpers"; +import { + formatNum, + isAutoresearchCommittableFile, + isAutoresearchLocalStatePath, + isAutoresearchShCommand, + normalizeAutoresearchPath, + readMaxExperiments, + readPendingRunSummary, + resolveWorkDir, + validateWorkDir, +} from "./helpers"; import promptTemplate from "./prompt.md" with { type: "text" }; import resumeMessageTemplate from "./resume-message.md" with { type: "text" }; import { cloneExperimentState, createExperimentState, createRuntimeStore, + currentResults, + findBaselineMetric, reconstructControlState, reconstructStateFromJsonl, } from "./state"; import { createInitExperimentTool } from "./tools/init-experiment"; import { createLogExperimentTool } from "./tools/log-experiment"; import { createRunExperimentTool } from "./tools/run-experiment"; -import type { AutoresearchRuntime } from "./types"; +import type { AutoresearchRuntime, ChecksResult, ExperimentResult, PendingRunSummary } from "./types"; -const AUTORESUME_INTERVAL_MS = 5 * 60 * 1000; -const MAX_AUTORESUME_TURNS = 20; const EXPERIMENT_TOOL_NAMES = ["init_experiment", "run_experiment", "log_experiment"]; +interface AutoresearchSetupInput { + intent: string; + benchmarkCommand: string; + metricName: string; + metricUnit: string; + direction: "lower" | "higher"; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; +} + export const createAutoresearchExtension: ExtensionFactory = api => { const runtimeStore = createRuntimeStore(); const dashboard = createDashboardController(); @@ -37,17 +59,18 @@ export const createAutoresearchExtension: ExtensionFactory = api => { const runtime = getRuntime(ctx); const workDir = resolveWorkDir(ctx.cwd); const reconstructed = reconstructStateFromJsonl(workDir); - const control = reconstructControlState(ctx.sessionManager.getEntries()); + const control = reconstructControlState(ctx.sessionManager.getBranch()); + const loggedRunNumbers = collectLoggedRunNumbers(reconstructed.state.results); runtime.state = cloneExperimentState(reconstructed.state); runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); runtime.goal = control.goal; runtime.autoresearchMode = control.autoresearchMode; - runtime.lastAutoResumeTime = 0; - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; - runtime.lastRunChecks = null; - runtime.lastRunDuration = null; - runtime.lastRunAsi = null; + runtime.lastRunSummary = await readPendingRunSummary(workDir, loggedRunNumbers); + runtime.lastRunChecks = summaryToChecks(runtime.lastRunSummary); + runtime.lastRunDuration = runtime.lastRunSummary?.durationSeconds ?? null; + runtime.lastRunAsi = runtime.lastRunSummary?.parsedAsi ?? null; + runtime.lastRunArtifactDir = runtime.lastRunSummary?.runDirectory ?? null; + runtime.lastRunNumber = runtime.lastRunSummary?.runNumber ?? null; runtime.runningExperiment = null; dashboard.updateWidget(ctx, runtime); const activeTools = api.getActiveTools(); @@ -78,6 +101,49 @@ export const createAutoresearchExtension: ExtensionFactory = api => { api.registerTool(createInitExperimentTool({ dashboard, getRuntime, pi: api })); api.registerTool(createRunExperimentTool({ dashboard, getRuntime, pi: api })); api.registerTool(createLogExperimentTool({ dashboard, getRuntime, pi: api })); + api.on("tool_call", (event, ctx) => { + const runtime = getRuntime(ctx); + if (!runtime.autoresearchMode) return; + if (event.toolName === "bash") { + const command = typeof event.input.command === "string" ? event.input.command : ""; + const validationError = validateAutoresearchBashCommand(command); + if (validationError) { + return { + block: true, + reason: validationError, + }; + } + return; + } + if (event.toolName !== "write" && event.toolName !== "edit" && event.toolName !== "ast_edit") return; + + const rawPaths = getGuardedToolPaths(event.toolName, event.input); + if (rawPaths === null) { + return { + block: true, + reason: + "Autoresearch requires an explicit target path for this editing tool so it can enforce Files in Scope and Off Limits before changes are made.", + }; + } + + const workDir = resolveWorkDir(ctx.cwd); + for (const rawPath of rawPaths) { + const relativePath = resolveAutoresearchRelativePath(workDir, rawPath); + if (!relativePath.ok) { + return { + block: true, + reason: relativePath.reason, + }; + } + const validationError = validateEditableAutoresearchPath(relativePath.relativePath, runtime); + if (validationError) { + return { + block: true, + reason: `Autoresearch blocked edits to ${relativePath.relativePath}: ${validationError}`, + }; + } + } + }); api.registerCommand("autoresearch", { description: "Start, stop, or clear builtin autoresearch mode.", @@ -102,8 +168,6 @@ export const createAutoresearchExtension: ExtensionFactory = api => { if (trimmed === "off") { setMode(ctx, false, runtime.goal, "off"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; dashboard.updateWidget(ctx, runtime); const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); @@ -113,34 +177,48 @@ export const createAutoresearchExtension: ExtensionFactory = api => { if (trimmed === "clear") { const workDir = resolveWorkDir(ctx.cwd); const jsonlPath = path.join(workDir, "autoresearch.jsonl"); + const localStatePath = path.join(workDir, ".autoresearch"); if (fs.existsSync(jsonlPath)) { fs.rmSync(jsonlPath); } + if (fs.existsSync(localStatePath)) { + fs.rmSync(localStatePath, { force: true, recursive: true }); + } runtime.state = createExperimentState(); runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); runtime.goal = null; + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = null; + runtime.lastRunNumber = null; + runtime.lastRunSummary = null; setMode(ctx, false, null, "clear"); dashboard.updateWidget(ctx, runtime); const experimentTools = new Set(EXPERIMENT_TOOL_NAMES); await api.setActiveTools(api.getActiveTools().filter(name => !experimentTools.has(name))); - ctx.ui.notify("Autoresearch log cleared", "info"); + ctx.ui.notify("Autoresearch local state cleared", "info"); return; } const workDir = resolveWorkDir(ctx.cwd); const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const hasAutoresearchMd = fs.existsSync(autoresearchMdPath); + const controlState = reconstructControlState(ctx.sessionManager.getBranch()); + const shouldResumeExistingNotes = + hasAutoresearchMd && + (hasLocalAutoresearchState(workDir) || (controlState.lastMode !== "clear" && trimmed.length === 0)); - if (hasAutoresearchMd) { - const branchResult = await ensureAutoresearchBranch(api, workDir, runtime.goal); + if (shouldResumeExistingNotes) { + const resumeContext = trimmed; + const resumeGoal = runtime.goal ?? runtime.state.name ?? null; + const branchResult = await ensureAutoresearchBranch(api, workDir, resumeGoal); if (!branchResult.ok) { ctx.ui.notify(branchResult.error, "error"); return; } - setMode(ctx, true, runtime.goal, "on"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; + setMode(ctx, true, resumeGoal, "on"); dashboard.updateWidget(ctx, runtime); await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); api.sendUserMessage( @@ -149,32 +227,34 @@ export const createAutoresearchExtension: ExtensionFactory = api => { branch_status_line: branchResult.created ? `Created and checked out dedicated git branch \`${branchResult.branchName}\` before resuming.` : `Using dedicated git branch \`${branchResult.branchName}\`.`, + has_resume_context: resumeContext.length > 0, + resume_context: resumeContext, }), ); return; } - const intentInput = await ctx.ui.input( - "Autoresearch Intent", + const setup = await promptForAutoresearchSetup( + ctx, trimmed || runtime.goal || "what should autoresearch improve?", ); - if (intentInput === undefined) return; + if (!setup) return; - const intent = intentInput.trim(); - if (intent.length === 0) { - ctx.ui.notify("Autoresearch intent is required", "info"); - return; - } - - const branchResult = await ensureAutoresearchBranch(api, workDir, intent); + const branchResult = await ensureAutoresearchBranch(api, workDir, setup.intent); if (!branchResult.ok) { ctx.ui.notify(branchResult.error, "error"); return; } - setMode(ctx, true, intent, "on"); - runtime.experimentsThisSession = 0; - runtime.autoResumeTurns = 0; + setMode(ctx, true, setup.intent, "on"); + runtime.state.name = setup.intent; + runtime.state.metricName = setup.metricName; + runtime.state.metricUnit = setup.metricUnit; + runtime.state.bestDirection = setup.direction; + runtime.state.benchmarkCommand = setup.benchmarkCommand; + runtime.state.scopePaths = [...setup.scopePaths]; + runtime.state.offLimits = [...setup.offLimits]; + runtime.state.constraints = [...setup.constraints]; dashboard.updateWidget(ctx, runtime); await api.setActiveTools([...new Set([...api.getActiveTools(), ...EXPERIMENT_TOOL_NAMES])]); api.sendUserMessage( @@ -182,7 +262,19 @@ export const createAutoresearchExtension: ExtensionFactory = api => { branch_status_line: branchResult.created ? `Created and checked out dedicated git branch \`${branchResult.branchName}\`.` : `Using dedicated git branch \`${branchResult.branchName}\`.`, - intent, + intent: setup.intent, + benchmark_command: setup.benchmarkCommand, + metric_name: setup.metricName, + metric_unit: setup.metricUnit, + direction: setup.direction, + scope_paths: setup.scopePaths, + scope_paths_block: formatBulletBlock(setup.scopePaths, value => ` - \`${value}\``), + has_off_limits: setup.offLimits.length > 0, + off_limits: setup.offLimits, + off_limits_block: formatBulletBlock(setup.offLimits, value => ` - \`${value}\``, " - `(none)`"), + has_constraints: setup.constraints.length > 0, + constraints: setup.constraints, + constraints_block: formatBulletBlock(setup.constraints, value => ` - ${value}`, " - `(none)`"), }), ); }, @@ -217,52 +309,358 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtimeStore.clear(getSessionKey(ctx)); }); - api.on("agent_start", (_event, ctx) => { - getRuntime(ctx).experimentsThisSession = 0; - }); - - api.on("agent_end", (_event, ctx) => { + api.on("agent_end", async (_event, ctx) => { const runtime = getRuntime(ctx); runtime.runningExperiment = null; dashboard.updateWidget(ctx, runtime); dashboard.requestRender(); if (!runtime.autoresearchMode) return; - if (runtime.experimentsThisSession === 0) return; - if (runtime.autoResumeTurns >= MAX_AUTORESUME_TURNS) return; - const now = Date.now(); - if (now - runtime.lastAutoResumeTime < AUTORESUME_INTERVAL_MS) return; - runtime.lastAutoResumeTime = now; - runtime.autoResumeTurns += 1; + if (ctx.hasPendingMessages()) return; const workDir = resolveWorkDir(ctx.cwd); + const pendingRun = + runtime.lastRunSummary ?? + (await readPendingRunSummary(workDir, collectLoggedRunNumbers(runtime.state.results))); + runtime.lastRunSummary = pendingRun; + runtime.lastRunChecks = summaryToChecks(pendingRun); + runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; + runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi; + const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const ideasPath = path.join(workDir, "autoresearch.ideas.md"); - api.sendUserMessage( - renderPromptTemplate(resumeMessageTemplate, { - has_ideas: fs.existsSync(ideasPath), - }), - { deliverAs: "followUp" }, + api.sendMessage( + { + customType: "autoresearch-resume", + content: renderPromptTemplate(resumeMessageTemplate, { + autoresearch_md_path: autoresearchMdPath, + has_ideas: fs.existsSync(ideasPath), + has_pending_run: Boolean(pendingRun), + }), + display: false, + attribution: "agent", + }, + { deliverAs: "nextTurn", triggerTurn: true }, ); }); - api.on("before_agent_start", (event, ctx) => { + api.on("before_agent_start", async (event, ctx) => { const runtime = getRuntime(ctx); if (!runtime.autoresearchMode) return; const workDir = resolveWorkDir(ctx.cwd); const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const checksPath = path.join(workDir, "autoresearch.checks.sh"); const ideasPath = path.join(workDir, "autoresearch.ideas.md"); + const programPath = path.join(workDir, "autoresearch.program.md"); + const pendingRun = + runtime.lastRunSummary ?? + (await readPendingRunSummary(workDir, collectLoggedRunNumbers(runtime.state.results))); + runtime.lastRunSummary = pendingRun; + runtime.lastRunChecks = summaryToChecks(pendingRun); + runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; + runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi; + const currentSegmentResults = currentResults(runtime.state.results, runtime.state.currentSegment); + const baselineMetric = findBaselineMetric(runtime.state.results, runtime.state.currentSegment); + const bestResult = findBestResult(runtime); + const goal = runtime.goal ?? runtime.state.name ?? ""; + const recentResults = currentSegmentResults.slice(-3).map(result => { + const asiSummary = summarizeExperimentAsi(result); + return { + asi_summary: asiSummary, + description: result.description, + has_asi_summary: Boolean(asiSummary), + metric_display: formatNum(result.metric, runtime.state.metricUnit), + run_number: result.runNumber ?? runtime.state.results.indexOf(result) + 1, + status: result.status, + }; + }); return { systemPrompt: renderPromptTemplate(promptTemplate, { base_system_prompt: event.systemPrompt, - goal: runtime.goal ?? event.prompt, + has_goal: goal.trim().length > 0, + goal, working_dir: workDir, default_metric_name: runtime.state.metricName, + metric_name: runtime.state.metricName, has_autoresearch_md: fs.existsSync(autoresearchMdPath), autoresearch_md_path: autoresearchMdPath, has_checks: fs.existsSync(checksPath), checks_path: checksPath, has_ideas: fs.existsSync(ideasPath), ideas_path: ideasPath, + has_program: fs.existsSync(programPath), + program_path: programPath, + current_segment: runtime.state.currentSegment + 1, + current_segment_run_count: currentSegmentResults.length, + has_baseline_metric: baselineMetric !== null, + baseline_metric_display: formatNum(baselineMetric, runtime.state.metricUnit), + has_best_result: Boolean(bestResult), + best_metric_display: bestResult + ? formatNum(bestResult.metric, runtime.state.metricUnit) + : formatNum(baselineMetric, runtime.state.metricUnit), + best_run_number: bestResult + ? (bestResult.runNumber ?? runtime.state.results.indexOf(bestResult) + 1) + : null, + has_recent_results: recentResults.length > 0, + recent_results: recentResults, + has_pending_run: Boolean(pendingRun), + pending_run_number: pendingRun?.runNumber, + pending_run_command: pendingRun?.command, + pending_run_directory: pendingRun?.runDirectory, + pending_run_passed: pendingRun?.passed ?? false, + has_pending_run_metric: pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined, + pending_run_metric_display: + pendingRun?.parsedPrimary !== null && pendingRun?.parsedPrimary !== undefined + ? formatNum(pendingRun.parsedPrimary, runtime.state.metricUnit) + : null, }), }; }); }; + +async function promptForAutoresearchSetup( + ctx: ExtensionContext, + defaultIntent: string, +): Promise { + const intentInput = await ctx.ui.input("Autoresearch Intent", defaultIntent); + if (intentInput === undefined) return undefined; + const intent = intentInput.trim(); + if (intent.length === 0) { + ctx.ui.notify("Autoresearch intent is required", "info"); + return undefined; + } + + const benchmarkCommandInput = await ctx.ui.input("Benchmark Command", "bash autoresearch.sh"); + if (benchmarkCommandInput === undefined) return undefined; + const benchmarkCommand = benchmarkCommandInput.trim(); + if (benchmarkCommand.length === 0) { + ctx.ui.notify("Benchmark command is required", "info"); + return undefined; + } + if (!isAutoresearchShCommand(benchmarkCommand)) { + ctx.ui.notify("Benchmark command must invoke `autoresearch.sh` directly", "info"); + return undefined; + } + + const metricNameInput = await ctx.ui.input("Primary Metric Name", "runtime_ms"); + if (metricNameInput === undefined) return undefined; + const metricName = metricNameInput.trim(); + if (metricName.length === 0) { + ctx.ui.notify("Primary metric name is required", "info"); + return undefined; + } + + const metricUnitInput = await ctx.ui.input("Metric Unit", "ms"); + if (metricUnitInput === undefined) return undefined; + const metricUnit = metricUnitInput.trim(); + + const directionInput = await ctx.ui.input("Metric Direction", "lower"); + if (directionInput === undefined) return undefined; + const normalizedDirection = directionInput.trim().toLowerCase(); + if (normalizedDirection !== "lower" && normalizedDirection !== "higher") { + ctx.ui.notify("Metric direction must be `lower` or `higher`", "info"); + return undefined; + } + + const scopePathsInput = await ctx.ui.input("Files in Scope", "packages/coding-agent/src/autoresearch"); + if (scopePathsInput === undefined) return undefined; + const scopePaths = splitSetupList(scopePathsInput); + if (scopePaths.length === 0) { + ctx.ui.notify("Files in Scope must include at least one path", "info"); + return undefined; + } + + const offLimitsInput = await ctx.ui.input("Off Limits", ""); + if (offLimitsInput === undefined) return undefined; + const constraintsInput = await ctx.ui.input("Constraints", ""); + if (constraintsInput === undefined) return undefined; + + return { + intent, + benchmarkCommand, + metricName, + metricUnit, + direction: normalizedDirection, + scopePaths, + offLimits: splitSetupList(offLimitsInput), + constraints: splitSetupList(constraintsInput), + }; +} + +function splitSetupList(value: string): string[] { + return value + .split(/\r?\n|,/) + .map(entry => entry.trim()) + .filter((entry, index, values) => entry.length > 0 && values.indexOf(entry) === index); +} + +function formatBulletBlock(values: string[], renderValue: (value: string) => string, emptyValue = ""): string { + if (values.length === 0) { + return emptyValue; + } + return values.map(renderValue).join("\n"); +} + +function hasLocalAutoresearchState(workDir: string): boolean { + return fs.existsSync(path.join(workDir, "autoresearch.jsonl")) || fs.existsSync(path.join(workDir, ".autoresearch")); +} + +function summarizeExperimentAsi(result: ExperimentResult): string | null { + const hypothesis = typeof result.asi?.hypothesis === "string" ? result.asi.hypothesis.trim() : ""; + const rollbackReason = typeof result.asi?.rollback_reason === "string" ? result.asi.rollback_reason.trim() : ""; + const nextActionHint = typeof result.asi?.next_action_hint === "string" ? result.asi.next_action_hint.trim() : ""; + const summary = [hypothesis, rollbackReason, nextActionHint].filter(part => part.length > 0).join(" | "); + return summary.length > 0 ? summary.slice(0, 220) : null; +} + +function getGuardedToolPaths(toolName: string, input: Record): string[] | null { + if (toolName === "write") { + return typeof input.path === "string" ? [input.path] : null; + } + if (toolName === "ast_edit") { + return typeof input.path === "string" ? [input.path] : null; + } + if (toolName !== "edit") { + return []; + } + + const paths: string[] = []; + if (typeof input.path === "string") { + paths.push(input.path); + } + if (typeof input.rename === "string") { + paths.push(input.rename); + } + if (typeof input.move === "string") { + paths.push(input.move); + } + return paths; +} + +function resolveAutoresearchRelativePath( + workDir: string, + rawPath: string, +): { ok: false; reason: string } | { ok: true; relativePath: string } { + if (looksLikeInternalUrl(rawPath)) { + return { + ok: false, + reason: `Autoresearch cannot validate internal URL paths during scoped editing: ${rawPath}`, + }; + } + const resolvedPath = path.isAbsolute(rawPath) ? path.resolve(rawPath) : path.resolve(workDir, rawPath); + const canonicalWorkDir = canonicalizeExistingPath(workDir); + const canonicalTargetPath = canonicalizeTargetPath(resolvedPath); + const relativePath = path.relative(canonicalWorkDir, canonicalTargetPath); + if (relativePath === ".." || relativePath.startsWith(`..${path.sep}`) || path.isAbsolute(relativePath)) { + return { + ok: false, + reason: `Autoresearch blocked edits outside the working tree: ${rawPath}`, + }; + } + return { + ok: true, + relativePath: relativePath.length === 0 ? "." : normalizeAutoresearchPath(relativePath), + }; +} + +function validateEditableAutoresearchPath(relativePath: string, runtime: AutoresearchRuntime): string | null { + if (isAutoresearchLocalStatePath(relativePath)) { + return "autoresearch local state files are managed by the experiment tools and cannot be edited directly"; + } + if (runtime.state.offLimits.some(spec => pathMatchesContractPath(relativePath, spec))) { + return "this path is listed under Off Limits in autoresearch.md"; + } + if (isAutoresearchCommittableFile(relativePath)) { + return null; + } + if (runtime.state.scopePaths.length === 0) { + return "Files in Scope is not initialized yet; only autoresearch control files may be edited before init_experiment runs"; + } + if (!runtime.state.scopePaths.some(spec => pathMatchesContractPath(relativePath, spec))) { + return "this path is outside Files in Scope in autoresearch.md"; + } + return null; +} + +function findBestResult(runtime: AutoresearchRuntime): ExperimentResult | null { + let best: ExperimentResult | null = null; + for (const result of runtime.state.results) { + if (result.segment !== runtime.state.currentSegment || result.status !== "keep") continue; + if (!best) { + best = result; + continue; + } + if (runtime.state.bestDirection === "lower" ? result.metric < best.metric : result.metric > best.metric) { + best = result; + } + } + return best; +} + +function collectLoggedRunNumbers(results: ExperimentResult[]): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} + +function summaryToChecks(summary: PendingRunSummary | null): ChecksResult | null { + if (!summary || summary.checksPass === null) { + return null; + } + return { + pass: summary.checksPass, + output: "", + duration: summary.checksDurationSeconds ?? 0, + }; +} + +function looksLikeInternalUrl(value: string): boolean { + return /^[a-z][a-z0-9+.-]*:\/\//i.test(value); +} + +function canonicalizeExistingPath(targetPath: string): string { + try { + return fs.realpathSync.native(targetPath); + } catch { + return path.resolve(targetPath); + } +} + +function canonicalizeTargetPath(targetPath: string): string { + const pendingSegments: string[] = []; + let currentPath = path.resolve(targetPath); + while (!fs.existsSync(currentPath)) { + const parentPath = path.dirname(currentPath); + if (parentPath === currentPath) { + return currentPath; + } + pendingSegments.unshift(path.basename(currentPath)); + currentPath = parentPath; + } + return path.resolve(canonicalizeExistingPath(currentPath), ...pendingSegments); +} + +function validateAutoresearchBashCommand(command: string): string | null { + const trimmed = command.trim(); + if (trimmed.length === 0) { + return null; + } + const mutationPatterns = [ + /(^|[;&|()]\s*)(?:bash|sh)\b/, + /(^|[;&|()]\s*)(?:python|python3|node|perl|ruby|php)\b/, + /(^|[;&|()]\s*)(?:mv|cp|rm|mkdir|touch|chmod|chown|ln|install|patch)\b/, + /(^|[;&|()]\s*)sed\s+-i\b/, + /(^|[;&|()]\s*)git\s+(?:add|apply|checkout|clean|commit|merge|rebase|reset|restore|revert|stash|switch|worktree)\b/, + /(^|[^<])>>?/, + /\|\s*tee\b/, + /<< pattern.test(trimmed))) { + return ( + "Autoresearch only allows read-only shell inspection. " + + "Use write/edit/ast_edit for file changes and run_experiment for benchmark execution." + ); + } + return null; +} diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md index 2ed34f189..c02c20f13 100644 --- a/packages/coding-agent/src/autoresearch/prompt.md +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -4,13 +4,60 @@ Autoresearch mode is active. +{{#if has_goal}} Primary goal: {{goal}} +{{else}} +Primary goal is documented in `autoresearch.md` for this session. +{{/if}} Working directory: `{{working_dir}}` You are running an autonomous experiment loop. Keep iterating until the user interrupts you or the configured maximum iteration count is reached. +{{#if has_program}} + +### Local Playbook + +`autoresearch.program.md` exists at `{{program_path}}`. + +Use it as a repo-local strategy overlay for this session. `autoresearch.md` remains the source of truth for benchmark, scope, and constraints. +{{/if}} +{{#if has_recent_results}} + +### Current Segment Snapshot + +- segment: `{{current_segment}}` +- runs in current segment: `{{current_segment_run_count}}` +{{#if has_baseline_metric}} +- baseline `{{metric_name}}`: `{{baseline_metric_display}}` +{{/if}} +{{#if has_best_result}} +- best kept `{{metric_name}}`: `{{best_metric_display}}`{{#if best_run_number}} from run `#{{best_run_number}}`{{/if}} +{{/if}} + +Recent runs: +{{#each recent_results}} +- run `#{{run_number}}`: `{{status}}` `{{metric_display}}` — {{description}} +{{#if has_asi_summary}} + ASI: {{asi_summary}} +{{/if}} +{{/each}} +{{/if}} +{{#if has_pending_run}} + +### Pending Run + +An unlogged run artifact exists at `{{pending_run_directory}}`. + +- run: `#{{pending_run_number}}` +- command: `{{pending_run_command}}` +{{#if has_pending_run_metric}} +- parsed `{{metric_name}}`: `{{pending_run_metric_display}}` +{{/if}} +- result status: {{#if pending_run_passed}}passed{{else}}failed{{/if}} +- finish the `log_experiment` step before starting another benchmark +{{/if}} ### Available tools @@ -80,12 +127,18 @@ Suggested structure: # Autoresearch ## Goal +{{#if has_goal}} - {{goal}} +{{else}} +- document the active target here before the first benchmark +{{/if}} ## Benchmark -- command: -- primary metric: -- secondary metrics: + - command: + - primary metric: + - metric unit: + - direction: + - secondary metrics: memory_mb, rss_mb ## Files in Scope - path: @@ -104,8 +157,9 @@ Suggested structure: - metric: - why it won: -## Ideas -- item +## What's Been Tried +- experiment: +- lesson: ``` ### Guardrails @@ -114,6 +168,7 @@ Suggested structure: - Do not overfit to synthetic inputs if the real workload is broader. - Preserve correctness. - Only modify files that are explicitly in scope for the current session. +- Do not use the general shell tool for file mutations during autoresearch. Use `write`, `edit`, or `ast_edit` for scoped code changes and `run_experiment` for benchmark execution. - If you create `autoresearch.checks.sh`, treat it as a hard gate for `keep`. - If the user sends another message while a run is in progress, finish the current run and logging cycle first, then address the new input in the next iteration. diff --git a/packages/coding-agent/src/autoresearch/resume-message.md b/packages/coding-agent/src/autoresearch/resume-message.md index 62c10b26a..64c4816f3 100644 --- a/packages/coding-agent/src/autoresearch/resume-message.md +++ b/packages/coding-agent/src/autoresearch/resume-message.md @@ -1,8 +1,13 @@ -The autoresearch loop ended unexpectedly. Resume it now. +Continue the autoresearch loop now. + +@{{autoresearch_md_path}} - Read `autoresearch.md` and `autoresearch.jsonl`. - Treat `autoresearch.md` as the source of truth for the current direction, scope, and constraints. - Inspect recent git history for context. +{{#if has_pending_run}} +- Inspect the latest unlogged `run.json` under `.autoresearch/runs/` and finish the pending `log_experiment` step before starting a new benchmark. +{{/if}} - Continue from the most promising unfinished direction. {{#if has_ideas}} - Review `autoresearch.ideas.md` for promising next steps and prune stale items. diff --git a/packages/coding-agent/src/autoresearch/state.ts b/packages/coding-agent/src/autoresearch/state.ts index 9d302d9b9..725a60715 100644 --- a/packages/coding-agent/src/autoresearch/state.ts +++ b/packages/coding-agent/src/autoresearch/state.ts @@ -1,6 +1,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import type { SessionEntry } from "../session/session-manager"; +import { normalizeAutoresearchList, normalizeContractPathSpec } from "./contract"; import { inferMetricUnitFromName, isBetter } from "./helpers"; import type { AutoresearchControlEntryData, @@ -29,6 +30,11 @@ export function createExperimentState(): ExperimentState { currentSegment: 0, maxExperiments: null, confidence: null, + benchmarkCommand: null, + scopePaths: [], + offLimits: [], + constraints: [], + segmentFingerprint: null, }; } @@ -36,12 +42,12 @@ export function createSessionRuntime(): AutoresearchRuntime { return { autoresearchMode: false, dashboardExpanded: false, - lastAutoResumeTime: 0, - experimentsThisSession: 0, - autoResumeTurns: 0, lastRunChecks: null, lastRunDuration: null, lastRunAsi: null, + lastRunArtifactDir: null, + lastRunNumber: null, + lastRunSummary: null, runningExperiment: null, state: createExperimentState(), goal: null, @@ -57,6 +63,9 @@ export function cloneExperimentState(state: ExperimentState): ExperimentState { asi: result.asi ? structuredClone(result.asi) : undefined, })), secondaryMetrics: state.secondaryMetrics.map(metric => ({ ...metric })), + scopePaths: [...state.scopePaths], + offLimits: [...state.offLimits], + constraints: [...state.constraints], }; } @@ -64,13 +73,35 @@ export function currentResults(results: ExperimentResult[], segment: number): Ex return results.filter(result => result.segment === segment); } +export function findBaselineResult(results: ExperimentResult[], segment: number): ExperimentResult | null { + return currentResults(results, segment).find(result => result.status === "keep") ?? null; +} + export function findBaselineMetric(results: ExperimentResult[], segment: number): number | null { - const baseline = results.find(result => result.segment === segment); + const baseline = findBaselineResult(results, segment); return baseline ? baseline.metric : null; } +export function findBestKeptMetric( + results: ExperimentResult[], + segment: number, + direction: MetricDirection, +): number | null { + let best: number | null = null; + for (const result of currentResults(results, segment)) { + if (result.status !== "keep") continue; + if (best === null || isBetter(result.metric, best, direction)) { + best = result.metric; + } + } + return best; +} + export function findBaselineRunNumber(results: ExperimentResult[], segment: number): number | null { - const index = results.findIndex(result => result.segment === segment); + const baseline = findBaselineResult(results, segment); + if (!baseline) return null; + if (baseline.runNumber !== null) return baseline.runNumber; + const index = results.indexOf(baseline); return index >= 0 ? index + 1 : null; } @@ -79,7 +110,7 @@ export function findBaselineSecondary( segment: number, knownMetrics: MetricDef[], ): NumericMetricMap { - const baseline = currentResults(results, segment)[0]; + const baseline = findBaselineResult(results, segment); const values: NumericMetricMap = baseline ? { ...baseline.metrics } : {}; for (const metric of knownMetrics) { if (values[metric.name] !== undefined) continue; @@ -155,22 +186,30 @@ export function reconstructStateFromJsonl(workDir: string): ReconstructedExperim continue; } - if (isConfigEntry(parsed)) { + const configEntry = parseConfigEntry(parsed); + if (configEntry) { if (sawConfig || state.results.length > 0) { segment += 1; } sawConfig = true; state.currentSegment = segment; - if (parsed.name) state.name = parsed.name; - if (parsed.metricName) state.metricName = parsed.metricName; - if (parsed.metricUnit !== undefined) state.metricUnit = parsed.metricUnit; - if (parsed.bestDirection) state.bestDirection = parsed.bestDirection; - state.secondaryMetrics = []; + if (configEntry.name) state.name = configEntry.name; + if (configEntry.metricName) state.metricName = configEntry.metricName; + if (configEntry.metricUnit !== undefined) state.metricUnit = configEntry.metricUnit; + if (configEntry.bestDirection) state.bestDirection = configEntry.bestDirection; + if (configEntry.benchmarkCommand !== undefined) state.benchmarkCommand = configEntry.benchmarkCommand; + state.scopePaths = cloneStringArray(configEntry.scopePaths); + state.offLimits = cloneStringArray(configEntry.offLimits); + state.constraints = cloneStringArray(configEntry.constraints); + state.segmentFingerprint = + typeof configEntry.segmentFingerprint === "string" ? configEntry.segmentFingerprint : null; + state.secondaryMetrics = hydrateMetricDefs(configEntry.secondaryMetrics); continue; } if (!isRunEntry(parsed)) continue; const result: ExperimentResult = { + runNumber: typeof parsed.run === "number" && Number.isFinite(parsed.run) ? parsed.run : null, commit: typeof parsed.commit === "string" ? parsed.commit : "", metric: typeof parsed.metric === "number" && Number.isFinite(parsed.metric) ? parsed.metric : 0, metrics: cloneNumericMetrics(parsed.metrics), @@ -195,17 +234,19 @@ export function reconstructStateFromJsonl(workDir: string): ReconstructedExperim export function reconstructControlState(entries: SessionEntry[]): ReconstructedControlState { let autoresearchMode = false; let goal: string | null = null; + let lastMode: ReconstructedControlState["lastMode"] = null; for (const entry of entries) { if (entry.type !== "custom" || entry.customType !== "autoresearch-control") continue; const data = parseControlEntry(entry.data); if (!data) continue; + lastMode = data.mode; autoresearchMode = data.mode === "on"; goal = data.goal ?? goal; if (data.mode === "clear") { goal = null; } } - return { autoresearchMode, goal }; + return { autoresearchMode, goal, lastMode }; } export function createRuntimeStore(): RuntimeStore { @@ -240,6 +281,51 @@ function isConfigEntry(value: unknown): value is AutoresearchJsonConfigEntry { return candidate.type === "config"; } +function parseConfigEntry(value: unknown): AutoresearchJsonConfigEntry | null { + if (!isConfigEntry(value)) return null; + const candidate = value as AutoresearchJsonConfigEntry; + const config: AutoresearchJsonConfigEntry = { type: "config" }; + if (typeof candidate.name === "string" && candidate.name.trim().length > 0) { + config.name = candidate.name; + } + if (typeof candidate.metricName === "string" && candidate.metricName.trim().length > 0) { + config.metricName = candidate.metricName; + } + if (typeof candidate.metricUnit === "string") { + config.metricUnit = candidate.metricUnit; + } + if (candidate.bestDirection === "lower" || candidate.bestDirection === "higher") { + config.bestDirection = candidate.bestDirection; + } + if (typeof candidate.benchmarkCommand === "string" && candidate.benchmarkCommand.trim().length > 0) { + config.benchmarkCommand = candidate.benchmarkCommand; + } + if (Array.isArray(candidate.secondaryMetrics)) { + config.secondaryMetrics = normalizeAutoresearchList( + candidate.secondaryMetrics.filter((item): item is string => typeof item === "string"), + ); + } + if (Array.isArray(candidate.scopePaths)) { + config.scopePaths = normalizeAutoresearchList( + candidate.scopePaths.filter((item): item is string => typeof item === "string").map(normalizeContractPathSpec), + ); + } + if (Array.isArray(candidate.offLimits)) { + config.offLimits = normalizeAutoresearchList( + candidate.offLimits.filter((item): item is string => typeof item === "string").map(normalizeContractPathSpec), + ); + } + if (Array.isArray(candidate.constraints)) { + config.constraints = normalizeAutoresearchList( + candidate.constraints.filter((item): item is string => typeof item === "string"), + ); + } + if (typeof candidate.segmentFingerprint === "string" && candidate.segmentFingerprint.trim().length > 0) { + config.segmentFingerprint = candidate.segmentFingerprint; + } + return config; +} + function isRunEntry(value: unknown): value is AutoresearchJsonRunEntry { if (typeof value !== "object" || value === null) return false; const candidate = value as { type?: unknown }; @@ -262,6 +348,19 @@ function cloneNumericMetrics(value: unknown): NumericMetricMap { return clone; } +function cloneStringArray(value: unknown): string[] { + if (!Array.isArray(value)) return []; + return value.filter((item): item is string => typeof item === "string"); +} + +function hydrateMetricDefs(metricNames: string[] | undefined): MetricDef[] { + if (!metricNames) return []; + return metricNames.map(name => ({ + name, + unit: inferMetricUnitFromName(name), + })); +} + function cloneAsi(value: unknown): ExperimentResult["asi"] { if (typeof value !== "object" || value === null) return undefined; return structuredClone(value) as ExperimentResult["asi"]; diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts index 19744fd2d..5e81714ba 100644 --- a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -5,7 +5,21 @@ import { Text } from "@oh-my-pi/pi-tui"; import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; -import { readMaxExperiments, resolveWorkDir, validateWorkDir } from "../helpers"; +import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; +import { + buildAutoresearchSegmentFingerprint, + contractListsEqual, + contractPathListsEqual, + loadAutoresearchScriptSnapshot, + readAutoresearchContract, +} from "../contract"; +import { + inferMetricUnitFromName, + isAutoresearchShCommand, + readMaxExperiments, + resolveWorkDir, + validateWorkDir, +} from "../helpers"; import { cloneExperimentState } from "../state"; import type { AutoresearchToolFactoryOptions, ExperimentState } from "../types"; @@ -26,6 +40,23 @@ const initExperimentSchema = Type.Object({ description: "Whether lower or higher values are better. Defaults to lower.", }), ), + benchmark_command: Type.String({ + description: "Benchmark command recorded in autoresearch.md.", + }), + scope_paths: Type.Array(Type.String(), { + description: "Files in Scope from autoresearch.md. Must be non-empty.", + minItems: 1, + }), + off_limits: Type.Optional( + Type.Array(Type.String(), { + description: "Off Limits paths from autoresearch.md.", + }), + ), + constraints: Type.Optional( + Type.Array(Type.String(), { + description: "Constraints from autoresearch.md.", + }), + ), }); interface InitExperimentDetails { @@ -53,6 +84,120 @@ export function createInitExperimentTool( const runtime = options.getRuntime(ctx); const state = runtime.state; const isReinitializing = state.results.length > 0; + const workDir = resolveWorkDir(ctx.cwd); + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + const errors = [...contractResult.errors, ...scriptSnapshot.errors]; + if (errors.length > 0) { + return { + content: [{ type: "text", text: `Error: ${errors.join(" ")}` }], + }; + } + + const benchmarkContract = contractResult.contract.benchmark; + const expectedDirection = benchmarkContract.direction ?? "lower"; + const expectedMetricUnit = benchmarkContract.metricUnit; + if (benchmarkContract.command && !isAutoresearchShCommand(benchmarkContract.command)) { + return { + content: [ + { + type: "text", + text: + "Error: Benchmark.command in autoresearch.md must invoke `autoresearch.sh` directly. " + + "Move the real workload into `autoresearch.sh` and re-run init_experiment.", + }, + ], + }; + } + if (benchmarkContract.command !== params.benchmark_command.trim()) { + return { + content: [ + { + type: "text", + text: + "Error: benchmark_command does not match autoresearch.md. " + + `Expected: ${benchmarkContract.command ?? "(missing)"}\nReceived: ${params.benchmark_command}`, + }, + ], + }; + } + if (benchmarkContract.primaryMetric !== params.metric_name.trim()) { + return { + content: [ + { + type: "text", + text: + "Error: metric_name does not match autoresearch.md. " + + `Expected: ${benchmarkContract.primaryMetric ?? "(missing)"}\nReceived: ${params.metric_name}`, + }, + ], + }; + } + if ((params.metric_unit ?? "") !== expectedMetricUnit) { + return { + content: [ + { + type: "text", + text: + "Error: metric_unit does not match autoresearch.md. " + + `Expected: ${expectedMetricUnit || "(empty)"}\nReceived: ${params.metric_unit ?? "(empty)"}`, + }, + ], + }; + } + if ((params.direction ?? "lower") !== expectedDirection) { + return { + content: [ + { + type: "text", + text: + "Error: direction does not match autoresearch.md. " + + `Expected: ${expectedDirection}\nReceived: ${params.direction ?? "lower"}`, + }, + ], + }; + } + if (!contractPathListsEqual(params.scope_paths, contractResult.contract.scopePaths)) { + return { + content: [ + { + type: "text", + text: + "Error: scope_paths do not match autoresearch.md. " + + `Expected: ${contractResult.contract.scopePaths.join(", ")}`, + }, + ], + }; + } + if (!contractPathListsEqual(params.off_limits ?? [], contractResult.contract.offLimits)) { + return { + content: [ + { + type: "text", + text: + "Error: off_limits do not match autoresearch.md. " + + `Expected: ${contractResult.contract.offLimits.join(", ") || "(empty)"}`, + }, + ], + }; + } + if (!contractListsEqual(params.constraints ?? [], contractResult.contract.constraints)) { + return { + content: [ + { + type: "text", + text: + "Error: constraints do not match autoresearch.md. " + + `Expected: ${contractResult.contract.constraints.join(", ") || "(empty)"}`, + }, + ], + }; + } + + const segmentFingerprint = buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); state.name = params.name; state.metricName = params.metric_name; @@ -61,12 +206,19 @@ export function createInitExperimentTool( state.maxExperiments = readMaxExperiments(ctx.cwd); state.bestMetric = null; state.confidence = null; - state.secondaryMetrics = []; + state.secondaryMetrics = benchmarkContract.secondaryMetrics.map(name => ({ + name, + unit: inferMetricUnitFromName(name), + })); + state.benchmarkCommand = params.benchmark_command.trim(); + state.scopePaths = [...contractResult.contract.scopePaths]; + state.offLimits = [...contractResult.contract.offLimits]; + state.constraints = [...contractResult.contract.constraints]; + state.segmentFingerprint = segmentFingerprint; if (isReinitializing) { state.currentSegment += 1; } - const workDir = resolveWorkDir(ctx.cwd); const jsonlPath = path.join(workDir, "autoresearch.jsonl"); const configLine = JSON.stringify({ type: "config", @@ -74,6 +226,12 @@ export function createInitExperimentTool( metricName: state.metricName, metricUnit: state.metricUnit, bestDirection: state.bestDirection, + benchmarkCommand: state.benchmarkCommand, + secondaryMetrics: state.secondaryMetrics.map(metric => metric.name), + scopePaths: state.scopePaths, + offLimits: state.offLimits, + constraints: state.constraints, + segmentFingerprint, }); if (isReinitializing) { @@ -89,7 +247,9 @@ export function createInitExperimentTool( const lines = [ `Experiment initialized: ${state.name}`, `Metric: ${state.metricName} (${state.metricUnit || "unitless"}, ${state.bestDirection} is better)`, + `Benchmark command: ${state.benchmarkCommand}`, `Working directory: ${workDir}`, + `Files in Scope: ${state.scopePaths.join(", ")}`, isReinitializing ? "Previous results remain in history. This starts a new segment and requires a fresh baseline." : "Now run the baseline experiment and log it.", @@ -107,12 +267,12 @@ export function createInitExperimentTool( return new Text(renderInitCall(args.name, theme), 0, 0); }, renderResult(result): Text { - const text = result.content.find(part => part.type === "text")?.text ?? ""; + const text = replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""); return new Text(text, 0, 0); }, }; } function renderInitCall(name: string, theme: Theme): string { - return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", name)}`; + return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", truncateToWidth(replaceTabs(name), 100))}`; } diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index 2a7420a46..1a25db2ef 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -2,14 +2,22 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { StringEnum } from "@oh-my-pi/pi-ai"; import { Text } from "@oh-my-pi/pi-tui"; +import { logger } from "@oh-my-pi/pi-utils"; import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; +import { replaceTabs, truncateToWidth } from "../../tools/render-utils"; +import { getAutoresearchFingerprintMismatchError, pathMatchesContractPath } from "../contract"; +import { getCurrentAutoresearchBranch, parseWorkDirDirtyPaths } from "../git"; import { + AUTORESEARCH_COMMITTABLE_FILES, formatNum, inferMetricUnitFromName, + isAutoresearchCommittableFile, + isAutoresearchLocalStatePath, + isBetter, mergeAsi, - PROTECTED_AUTORESEARCH_FILES, + readPendingRunSummary, resolveWorkDir, validateWorkDir, } from "../helpers"; @@ -19,6 +27,7 @@ import { currentResults, findBaselineMetric, findBaselineSecondary, + findBestKeptMetric, } from "../state"; import type { ASIData, @@ -29,6 +38,8 @@ import type { NumericMetricMap, } from "../types"; +const EXPERIMENT_TOOL_NAMES = ["init_experiment", "run_experiment", "log_experiment"]; + const logExperimentSchema = Type.Object({ commit: Type.String({ description: "Current git commit hash or placeholder.", @@ -64,6 +75,11 @@ interface PreservedFile { path: string; } +interface KeepCommitResult { + error?: string; + note?: string; +} + export function createLogExperimentTool( options: AutoresearchToolFactoryOptions, ): ToolDefinition { @@ -85,7 +101,55 @@ export function createLogExperimentTool( const runtime = options.getRuntime(ctx); const state = runtime.state; const workDir = resolveWorkDir(ctx.cwd); - const secondaryMetrics = cloneMetrics(params.metrics); + const fingerprintError = getAutoresearchFingerprintMismatchError(state.segmentFingerprint, workDir); + if (fingerprintError) { + return { + content: [{ type: "text", text: `Error: ${fingerprintError}` }], + }; + } + + const pendingRun = + runtime.lastRunSummary ?? (await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results))); + if (!pendingRun) { + return { + content: [{ type: "text", text: "Error: no unlogged run is available. Run run_experiment first." }], + }; + } + runtime.lastRunSummary = pendingRun; + runtime.lastRunAsi = pendingRun.parsedAsi; + runtime.lastRunChecks = + pendingRun.checksPass === null + ? null + : { + pass: pendingRun.checksPass, + output: "", + duration: pendingRun.checksDurationSeconds ?? 0, + }; + runtime.lastRunDuration = pendingRun.durationSeconds; + + if (pendingRun.parsedPrimary !== null && params.metric !== pendingRun.parsedPrimary) { + return { + content: [ + { + type: "text", + text: + "Error: metric does not match the parsed primary metric from the pending run.\n" + + `Expected: ${pendingRun.parsedPrimary}\nReceived: ${params.metric}`, + }, + ], + }; + } + + if (params.status === "keep" && !pendingRun.passed) { + return { + content: [ + { + type: "text", + text: "Error: cannot keep this run because the pending benchmark did not pass. Log it as crash or checks_failed instead.", + }, + ], + }; + } if (params.status === "keep" && runtime.lastRunChecks && !runtime.lastRunChecks.pass) { return { @@ -98,6 +162,14 @@ export function createLogExperimentTool( }; } + const observedStatusError = validateObservedStatus(params.status, pendingRun); + if (observedStatusError) { + return { + content: [{ type: "text", text: `Error: ${observedStatusError}` }], + }; + } + + const secondaryMetrics = buildSecondaryMetrics(params.metrics, pendingRun.parsedMetrics, state.metricName); const validationError = validateSecondaryMetrics(state, secondaryMetrics, params.force ?? false); if (validationError) { return { @@ -112,7 +184,37 @@ export function createLogExperimentTool( content: [{ type: "text", text: `Error: ${asiValidationError}` }], }; } + + let keepScopeValidation: { committablePaths: string[] } | undefined; + if (params.status === "keep") { + const scopeValidation = await validateKeepPaths(options, workDir, state); + if (typeof scopeValidation === "string") { + return { + content: [{ type: "text", text: `Error: ${scopeValidation}` }], + }; + } + const currentBestMetric = findBestKeptMetric(state.results, state.currentSegment, state.bestDirection); + if ( + currentBestMetric !== null && + params.metric !== currentBestMetric && + !isBetter(params.metric, currentBestMetric, state.bestDirection) + ) { + return { + content: [ + { + type: "text", + text: + "Error: cannot keep this run because the primary metric regressed.\n" + + `Current best: ${currentBestMetric}\nReceived: ${params.metric}`, + }, + ], + }; + } + keepScopeValidation = scopeValidation; + } + const experiment: ExperimentResult = { + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, commit: params.commit.slice(0, 7), metric: params.metric, metrics: secondaryMetrics, @@ -124,32 +226,96 @@ export function createLogExperimentTool( asi: mergedAsi, }; + const activeBranch = await getCurrentAutoresearchBranch(options.pi, workDir); + if (!activeBranch) { + return { + content: [ + { + type: "text", + text: + "Error: autoresearch keep/discard actions require an active `autoresearch/...` branch. " + + "Run `/autoresearch` again to restore the protected branch before logging this run.", + }, + ], + }; + } + + let gitNote: string | null = null; + if (params.status === "keep") { + const commitResult = await commitKeptExperiment(options, workDir, state, experiment, keepScopeValidation); + if (commitResult.error) { + return { + content: [{ type: "text", text: `Error: ${commitResult.error}` }], + }; + } + gitNote = commitResult.note ?? null; + } else { + const revertResult = await revertFailedExperiment(options, workDir); + if (revertResult.error) { + return { + content: [{ type: "text", text: `Error: ${revertResult.error}` }], + }; + } + gitNote = revertResult.note ?? null; + } + + const previousState = cloneExperimentState(state); state.results.push(experiment); - runtime.experimentsThisSession += 1; registerSecondaryMetrics(state, secondaryMetrics); state.bestMetric = findBaselineMetric(state.results, state.currentSegment); state.confidence = computeConfidence(state.results, state.currentSegment, state.bestDirection); experiment.confidence = state.confidence; - persistRun(workDir, state.results.length, experiment); - - let gitNote: string | null = null; - if (params.status === "keep") { - gitNote = await commitKeptExperiment(options, workDir, state, experiment); - } else { - gitNote = await revertFailedExperiment(options, workDir); + const wallClockSeconds = runtime.lastRunDuration; + try { + persistRun(workDir, experiment); + } catch (error) { + runtime.state = previousState; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); + throw error; + } + try { + await updateRunMetadata(runtime.lastRunArtifactDir ?? pendingRun.runDirectory, { + commit: experiment.commit, + confidence: experiment.confidence, + description: experiment.description, + gitNote, + loggedAt: new Date(experiment.timestamp).toISOString(), + loggedAsi: experiment.asi, + loggedMetric: experiment.metric, + loggedMetrics: experiment.metrics, + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, + status: experiment.status, + wallClockSeconds, + }); + } catch (error) { + logger.warn("Failed to update autoresearch run metadata after persisting JSONL history", { + error: error instanceof Error ? error.message : String(error), + runDirectory: runtime.lastRunArtifactDir ?? pendingRun.runDirectory, + runNumber: runtime.lastRunNumber ?? pendingRun.runNumber, + }); } - const wallClockSeconds = runtime.lastRunDuration; runtime.runningExperiment = null; runtime.lastRunChecks = null; runtime.lastRunDuration = null; runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = null; + runtime.lastRunNumber = null; + runtime.lastRunSummary = null; const currentSegmentRuns = currentResults(state.results, state.currentSegment).length; const text = buildLogText(state, experiment, currentSegmentRuns, wallClockSeconds, gitNote); if (state.maxExperiments !== null && currentSegmentRuns >= state.maxExperiments) { runtime.autoresearchMode = false; + options.pi.appendEntry( + "autoresearch-control", + runtime.goal ? { mode: "off", goal: runtime.goal } : { mode: "off" }, + ); + await options.pi.setActiveTools( + options.pi.getActiveTools().filter(name => !EXPERIMENT_TOOL_NAMES.includes(name)), + ); } options.dashboard.updateWidget(ctx, runtime); options.dashboard.requestRender(); @@ -169,8 +335,9 @@ export function createLogExperimentTool( }, renderCall(args, _options, theme): Text { const color = args.status === "keep" ? "success" : args.status === "discard" ? "warning" : "error"; + const description = truncateToWidth(replaceTabs(args.description), 100); return new Text( - `${theme.fg("toolTitle", theme.bold("log_experiment"))} ${theme.fg(color, args.status)} ${theme.fg("muted", args.description)}`, + `${theme.fg("toolTitle", theme.bold("log_experiment"))} ${theme.fg(color, args.status)} ${theme.fg("muted", description)}`, 0, 0, ); @@ -178,7 +345,7 @@ export function createLogExperimentTool( renderResult(result, _options, theme): Text { const details = result.details; if (!details) { - return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + return new Text(replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""), 0, 0); } const summary = renderSummary(details, theme); return new Text(summary, 0, 0); @@ -190,6 +357,22 @@ function cloneMetrics(value: NumericMetricMap | undefined): NumericMetricMap { return value ? { ...value } : {}; } +function buildSecondaryMetrics( + overrides: NumericMetricMap | undefined, + parsedMetrics: NumericMetricMap | null, + primaryMetricName: string, +): NumericMetricMap { + const merged: NumericMetricMap = {}; + for (const [name, value] of Object.entries(parsedMetrics ?? {})) { + if (name === primaryMetricName) continue; + merged[name] = value; + } + for (const [name, value] of Object.entries(cloneMetrics(overrides))) { + merged[name] = value; + } + return merged; +} + function sanitizeAsi(value: { [key: string]: unknown } | undefined): ASIData | undefined { if (!value) return undefined; const result: ASIData = {}; @@ -269,29 +452,71 @@ function registerSecondaryMetrics(state: ExperimentState, metrics: NumericMetric } } -function persistRun(workDir: string, runNumber: number, experiment: ExperimentResult): void { +function persistRun(workDir: string, experiment: ExperimentResult): void { const entry = { - run: runNumber, + run: experiment.runNumber, ...experiment, }; const jsonlPath = path.join(workDir, "autoresearch.jsonl"); fs.appendFileSync(jsonlPath, `${JSON.stringify(entry)}\n`); } +function collectLoggedRunNumbers(results: ExperimentResult[]): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} + +function validateObservedStatus( + status: ExperimentResult["status"], + pendingRun: { checksPass: boolean | null; passed: boolean }, +): string | null { + if (pendingRun.checksPass === false) { + return status === "checks_failed" + ? null + : "benchmark checks failed for the pending run. Log it as checks_failed."; + } + if (!pendingRun.passed) { + return status === "crash" ? null : "the pending benchmark failed. Log it as crash."; + } + return status === "keep" || status === "discard" ? null : "the pending benchmark passed. Log it as keep or discard."; +} + async function commitKeptExperiment( options: AutoresearchToolFactoryOptions, workDir: string, state: ExperimentState, experiment: ExperimentResult, -): Promise { - const addResult = await options.pi.exec("git", ["add", "-A"], { cwd: workDir, timeout: 10_000 }); - if (addResult.code !== 0) { - return `git add failed: ${mergeStdoutStderr(addResult).trim() || `exit ${addResult.code}`}`; + scopeValidation: { committablePaths: string[] } | undefined, +): Promise { + if (!scopeValidation || scopeValidation.committablePaths.length === 0) { + return { note: "nothing to commit" }; } - const diffResult = await options.pi.exec("git", ["diff", "--cached", "--quiet"], { cwd: workDir, timeout: 10_000 }); + const addResult = await options.pi.exec("git", ["add", "--all", "--", ...scopeValidation.committablePaths], { + cwd: workDir, + timeout: 10_000, + }); + if (addResult.code !== 0) { + return { + error: `git add failed: ${mergeStdoutStderr(addResult).trim() || `exit ${addResult.code}`}`, + }; + } + + const diffResult = await options.pi.exec( + "git", + ["diff", "--cached", "--quiet", "--", ...scopeValidation.committablePaths], + { + cwd: workDir, + timeout: 10_000, + }, + ); if (diffResult.code === 0) { - return "nothing to commit"; + return { note: "nothing to commit" }; } const payload: { [key: string]: string | number } = { @@ -302,12 +527,18 @@ async function commitKeptExperiment( payload[name] = value; } const commitMessage = `${experiment.description}\n\nResult: ${JSON.stringify(payload)}`; - const commitResult = await options.pi.exec("git", ["commit", "-m", commitMessage], { - cwd: workDir, - timeout: 10_000, - }); + const commitResult = await options.pi.exec( + "git", + ["commit", "-m", commitMessage, "--", ...scopeValidation.committablePaths], + { + cwd: workDir, + timeout: 10_000, + }, + ); if (commitResult.code !== 0) { - return `git commit failed: ${mergeStdoutStderr(commitResult).trim() || `exit ${commitResult.code}`}`; + return { + error: `git commit failed: ${mergeStdoutStderr(commitResult).trim() || `exit ${commitResult.code}`}`, + }; } const revParseResult = await options.pi.exec("git", ["rev-parse", "--short=7", "HEAD"], { @@ -322,28 +553,58 @@ async function commitKeptExperiment( mergeStdoutStderr(commitResult) .split("\n") .find(line => line.trim().length > 0) ?? "committed"; - return summaryLine.trim(); + return { note: summaryLine.trim() }; } -async function revertFailedExperiment(options: AutoresearchToolFactoryOptions, workDir: string): Promise { +async function revertFailedExperiment( + options: AutoresearchToolFactoryOptions, + workDir: string, +): Promise { const preservedFiles = preserveAutoresearchFiles(workDir); - const resetResult = await options.pi.exec("git", ["reset", "--hard", "HEAD"], { cwd: workDir, timeout: 10_000 }); - const cleanResult = await options.pi.exec("git", ["clean", "-fd"], { cwd: workDir, timeout: 10_000 }); + const restoreResult = await options.pi.exec( + "git", + ["restore", "--source=HEAD", "--staged", "--worktree", "--", "."], + { cwd: workDir, timeout: 10_000 }, + ); + const cleanResult = await options.pi.exec("git", ["clean", "-fd", "--", "."], { cwd: workDir, timeout: 10_000 }); restoreAutoresearchFiles(preservedFiles); - - const notes: string[] = ["reverted changes"]; - if (resetResult.code !== 0) { - notes.push(`git reset failed: ${mergeStdoutStderr(resetResult).trim() || `exit ${resetResult.code}`}`); + if (restoreResult.code !== 0) { + return { + error: `git restore failed: ${mergeStdoutStderr(restoreResult).trim() || `exit ${restoreResult.code}`}`, + }; } if (cleanResult.code !== 0) { - notes.push(`git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`); + return { + error: `git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`, + }; } - return notes.join("; "); + const dirtyCheckResult = await options.pi.exec( + "git", + ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."], + { cwd: workDir, timeout: 10_000 }, + ); + if (dirtyCheckResult.code !== 0) { + return { + error: `git status failed after cleanup: ${mergeStdoutStderr(dirtyCheckResult).trim() || `exit ${dirtyCheckResult.code}`}`, + }; + } + const workDirPrefix = await readGitWorkDirPrefix(options, workDir); + const remainingDirtyPaths = parseWorkDirDirtyPaths(dirtyCheckResult.stdout, workDirPrefix).filter( + relativePath => !isAutoresearchLocalStatePath(relativePath), + ); + if (remainingDirtyPaths.length > 0) { + return { + error: + "Autoresearch cleanup left the worktree dirty. Resolve these paths before continuing: " + + remainingDirtyPaths.join(", "), + }; + } + return { note: "reverted changes" }; } function preserveAutoresearchFiles(workDir: string): PreservedFile[] { const files: PreservedFile[] = []; - for (const relativePath of PROTECTED_AUTORESEARCH_FILES) { + for (const relativePath of [...AUTORESEARCH_COMMITTABLE_FILES, "autoresearch.jsonl"]) { const absolutePath = path.join(workDir, relativePath); if (!fs.existsSync(absolutePath)) continue; files.push({ @@ -351,6 +612,10 @@ function preserveAutoresearchFiles(workDir: string): PreservedFile[] { path: absolutePath, }); } + const localStateDir = path.join(workDir, ".autoresearch"); + if (fs.existsSync(localStateDir)) { + collectDirectoryFiles(localStateDir, files); + } return files; } @@ -365,6 +630,110 @@ function mergeStdoutStderr(result: { stderr: string; stdout: string }): string { return `${result.stdout}${result.stderr}`; } +async function validateKeepPaths( + options: AutoresearchToolFactoryOptions, + workDir: string, + state: ExperimentState, +): Promise<{ committablePaths: string[] } | string> { + if (state.scopePaths.length === 0) { + return "Files in Scope is empty for the current segment. Re-run init_experiment after fixing autoresearch.md."; + } + + const statusResult = await options.pi.exec( + "git", + ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."], + { + cwd: workDir, + timeout: 10_000, + }, + ); + if (statusResult.code !== 0) { + return `git status failed: ${mergeStdoutStderr(statusResult).trim() || `exit ${statusResult.code}`}`; + } + + const workDirPrefix = await readGitWorkDirPrefix(options, workDir); + const committablePaths: string[] = []; + for (const normalizedPath of parseWorkDirDirtyPaths(statusResult.stdout, workDirPrefix)) { + if (isAutoresearchLocalStatePath(normalizedPath)) { + continue; + } + if (isAutoresearchCommittableFile(normalizedPath)) { + committablePaths.push(normalizedPath); + continue; + } + if (state.offLimits.some(spec => pathMatchesContractPath(normalizedPath, spec))) { + return `cannot keep this run because ${normalizedPath} is listed under Off Limits in autoresearch.md`; + } + if (!state.scopePaths.some(spec => pathMatchesContractPath(normalizedPath, spec))) { + return `cannot keep this run because ${normalizedPath} is outside Files in Scope`; + } + committablePaths.push(normalizedPath); + } + + return { committablePaths }; +} + +function collectDirectoryFiles(directory: string, files: PreservedFile[]): void { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const absolutePath = path.join(directory, entry.name); + if (entry.isDirectory()) { + collectDirectoryFiles(absolutePath, files); + continue; + } + files.push({ + content: fs.readFileSync(absolutePath), + path: absolutePath, + }); + } +} + +async function updateRunMetadata( + runDirectory: string | null, + metadata: { + commit: string; + confidence: number | null; + description: string; + gitNote: string | null; + loggedAt: string; + loggedAsi: ASIData | undefined; + loggedMetric: number; + loggedMetrics: NumericMetricMap; + runNumber: number | null; + status: ExperimentResult["status"]; + wallClockSeconds: number | null; + }, +): Promise { + if (!runDirectory) return; + const runJsonPath = path.join(runDirectory, "run.json"); + let existing: Record = {}; + try { + existing = (await Bun.file(runJsonPath).json()) as Record; + } catch { + existing = {}; + } + await Bun.write( + runJsonPath, + JSON.stringify( + { + ...existing, + loggedRunNumber: metadata.runNumber, + loggedAt: metadata.loggedAt, + loggedAsi: metadata.loggedAsi, + loggedMetric: metadata.loggedMetric, + loggedMetrics: metadata.loggedMetrics, + status: metadata.status, + description: metadata.description, + commit: metadata.commit, + gitNote: metadata.gitNote, + confidence: metadata.confidence, + wallClockSeconds: metadata.wallClockSeconds, + }, + null, + 2, + ), + ); +} + function buildLogText( state: ExperimentState, experiment: ExperimentResult, @@ -372,7 +741,8 @@ function buildLogText( wallClockSeconds: number | null, gitNote: string | null, ): string { - const lines = [`Logged run #${state.results.length}: ${experiment.status} - ${experiment.description}`]; + const displayRunNumber = experiment.runNumber ?? state.results.length; + const lines = [`Logged run #${displayRunNumber}: ${experiment.status} - ${experiment.description}`]; if (wallClockSeconds !== null) { lines.push(`Wall clock: ${wallClockSeconds.toFixed(1)}s`); } @@ -422,6 +792,12 @@ function buildLogText( return lines.join("\n"); } +async function readGitWorkDirPrefix(options: AutoresearchToolFactoryOptions, workDir: string): Promise { + const prefixResult = await options.pi.exec("git", ["rev-parse", "--show-prefix"], { cwd: workDir, timeout: 5_000 }); + if (prefixResult.code !== 0) return ""; + return prefixResult.stdout.trim(); +} + function truncateAsiValue(value: ASIData[string]): string { const text = typeof value === "string" ? value : JSON.stringify(value); return text.length > 120 ? `${text.slice(0, 117)}...` : text; @@ -430,7 +806,7 @@ function truncateAsiValue(value: ASIData[string]): string { function renderSummary(details: LogDetails, theme: Theme): string { const { experiment, state } = details; const color = experiment.status === "keep" ? "success" : experiment.status === "discard" ? "warning" : "error"; - let summary = `${theme.fg(color, experiment.status.toUpperCase())} ${theme.fg("muted", experiment.description)}`; + let summary = `${theme.fg(color, experiment.status.toUpperCase())} ${theme.fg("muted", truncateToWidth(replaceTabs(experiment.description), 100))}`; summary += ` ${theme.fg("accent", `${state.metricName}=${formatNum(experiment.metric, state.metricUnit)}`)}`; if (state.bestMetric !== null) { summary += ` ${theme.fg("dim", `baseline ${formatNum(state.bestMetric, state.metricUnit)}`)}`; diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts index a5e502b15..a393c3647 100644 --- a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -7,16 +7,20 @@ import { Type } from "@sinclair/typebox"; import type { ToolDefinition } from "../../extensibility/extensions"; import type { Theme } from "../../modes/theme/theme"; import { DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncateTail } from "../../session/streaming-output"; +import { replaceTabs, shortenPath, truncateToWidth } from "../../tools/render-utils"; +import { getAutoresearchFingerprintMismatchError } from "../contract"; import { - createTempFileAllocator, EXPERIMENT_MAX_BYTES, EXPERIMENT_MAX_LINES, formatElapsed, formatNum, + getAutoresearchRunDirectory, + getNextAutoresearchRunNumber, isAutoresearchShCommand, killTree, parseAsiLines, parseMetricLines, + readPendingRunSummary, resolveWorkDir, validateWorkDir, } from "../helpers"; @@ -39,19 +43,27 @@ const runExperimentSchema = Type.Object({ }); interface ProcessExecutionResult { - actualTotalBytes: number; exitCode: number | null; killed: boolean; + logPath: string; output: string; - tempFilePath?: string; } interface ChecksExecutionResult { code: number | null; killed: boolean; + logPath: string; output: string; } +interface ProgressSnapshot { + elapsed: string; + runDirectory: string; + fullOutputPath: string; + tailOutput: string; + truncation?: RunExperimentProgressDetails["truncation"]; +} + export function createRunExperimentTool( options: AutoresearchToolFactoryOptions, ): ToolDefinition { @@ -59,7 +71,7 @@ export function createRunExperimentTool( name: "run_experiment", label: "Run Experiment", description: - "Run an experiment command with timing, tail capture, structured metric parsing, and optional autoresearch.checks.sh validation.", + "Run an experiment command with timing, output capture, structured metric parsing, durable run artifacts, and optional autoresearch.checks.sh validation.", parameters: runExperimentSchema, defaultInactive: true, async execute(_toolCallId, params, signal, onUpdate, ctx) { @@ -75,6 +87,25 @@ export function createRunExperimentTool( const workDir = resolveWorkDir(ctx.cwd); const checksPath = path.join(workDir, "autoresearch.checks.sh"); const autoresearchScriptPath = path.join(workDir, "autoresearch.sh"); + const fingerprintError = getAutoresearchFingerprintMismatchError(state.segmentFingerprint, workDir); + if (fingerprintError) { + return { + content: [{ type: "text", text: `Error: ${fingerprintError}` }], + }; + } + + if (state.benchmarkCommand && params.command.trim() !== state.benchmarkCommand) { + return { + content: [ + { + type: "text", + text: + "Error: command does not match the benchmark command recorded for this segment.\n" + + `Expected: ${state.benchmarkCommand}\nReceived: ${params.command}`, + }, + ], + }; + } if (fs.existsSync(autoresearchScriptPath) && !isAutoresearchShCommand(params.command)) { return { @@ -104,9 +135,54 @@ export function createRunExperimentTool( } } + const pendingRun = + runtime.lastRunSummary ?? (await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results))); + if (pendingRun) { + return { + content: [ + { + type: "text", + text: + `Error: run #${pendingRun.runNumber} has not been logged yet. ` + + "Call log_experiment before starting another benchmark run.", + }, + ], + }; + } + + const runNumber = getNextAutoresearchRunNumber(workDir, runtime.lastRunNumber); + const runDirectory = getAutoresearchRunDirectory(workDir, runNumber); + const benchmarkLogPath = path.join(runDirectory, "benchmark.log"); + const checksLogPath = path.join(runDirectory, "checks.log"); + const runJsonPath = path.join(runDirectory, "run.json"); + await fs.promises.mkdir(runDirectory, { recursive: true }); + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = runDirectory; + runtime.lastRunNumber = runNumber; + runtime.lastRunSummary = null; + await Bun.write( + runJsonPath, + JSON.stringify( + { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath, + command: params.command, + startedAt: new Date().toISOString(), + }, + null, + 2, + ), + ); + runtime.runningExperiment = { startedAt: Date.now(), command: params.command, + runDirectory, + runNumber, }; options.dashboard.updateWidget(ctx, runtime); options.dashboard.requestRender(); @@ -116,8 +192,9 @@ export function createRunExperimentTool( let execution: ProcessExecutionResult; try { execution = await executeProcess({ - command: params.command, + command: ["bash", "-lc", params.command], cwd: workDir, + logPath: benchmarkLogPath, timeoutMs, signal, onProgress: details => { @@ -128,6 +205,7 @@ export function createRunExperimentTool( elapsed: details.elapsed, truncation: details.truncation, fullOutputPath: details.fullOutputPath, + runDirectory: details.runDirectory, }, }); }, @@ -146,12 +224,14 @@ export function createRunExperimentTool( let checksTimedOut = false; let checksOutput = ""; let checksDuration = 0; + let checksLogPathValue: string | undefined; if (benchmarkPassed && fs.existsSync(checksPath)) { const checksStartedAt = Date.now(); - const checksResult = runChecks({ + const checksResult = await runChecks({ cwd: workDir, pathToChecks: checksPath, + logPath: checksLogPath, timeoutMs: Math.max(0, Math.floor((params.checks_timeout_seconds ?? 300) * 1000)), signal, }); @@ -159,6 +239,7 @@ export function createRunExperimentTool( checksTimedOut = checksResult.killed; checksPass = checksResult.code === 0 && !checksResult.killed; checksOutput = checksResult.output; + checksLogPathValue = checksResult.logPath; } runtime.lastRunChecks = @@ -179,12 +260,6 @@ export function createRunExperimentTool( maxLines: DEFAULT_MAX_LINES, }); - let fullOutputPath = execution.tempFilePath; - if (!fullOutputPath && llmTruncation.truncated) { - fullOutputPath = createTempFileAllocator()(); - fs.writeFileSync(fullOutputPath, execution.output); - } - const parsedMetricsMap = parseMetricLines(execution.output); const parsedMetrics = parsedMetricsMap.size > 0 ? Object.fromEntries(parsedMetricsMap.entries()) : null; const parsedPrimary = parsedMetricsMap.get(state.metricName) ?? null; @@ -192,6 +267,10 @@ export function createRunExperimentTool( runtime.lastRunAsi = parsedAsi; const resultDetails: RunDetails = { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath: checksLogPathValue, command: params.command, exitCode: execution.exitCode, durationSeconds, @@ -209,8 +288,50 @@ export function createRunExperimentTool( metricName: state.metricName, metricUnit: state.metricUnit, truncation: llmTruncation.truncated ? llmTruncation : undefined, - fullOutputPath, + fullOutputPath: execution.logPath, }; + runtime.lastRunSummary = { + checksDurationSeconds: checksDuration, + checksPass, + checksTimedOut, + command: params.command, + durationSeconds, + parsedAsi, + parsedMetrics, + parsedPrimary, + passed: resultDetails.passed, + runDirectory, + runNumber, + }; + + await Bun.write( + runJsonPath, + JSON.stringify( + { + runNumber, + runDirectory, + benchmarkLogPath, + checksLogPath: checksLogPathValue, + command: params.command, + completedAt: new Date().toISOString(), + durationSeconds, + exitCode: execution.exitCode, + timedOut: execution.killed, + checks: { + durationSeconds: checksDuration, + passed: checksPass, + timedOut: checksTimedOut, + }, + parsedMetrics, + parsedPrimary, + parsedAsi, + truncation: resultDetails.truncation, + fullOutputPath: resultDetails.fullOutputPath, + }, + null, + 2, + ), + ); return { content: [{ type: "text", text: buildRunText(resultDetails, llmTruncation.content, state.bestMetric) }], @@ -218,8 +339,9 @@ export function createRunExperimentTool( }; }, renderCall(args, _options, theme): Text { + const commandPreview = truncateToWidth(replaceTabs(args.command), 100); return new Text( - `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", args.command)}`, + `${theme.fg("toolTitle", theme.bold("run_experiment"))} ${theme.fg("muted", commandPreview)}`, 0, 0, ); @@ -227,13 +349,13 @@ export function createRunExperimentTool( renderResult(result, options, theme): Text { if (isProgressDetails(result.details)) { const header = theme.fg("warning", `Running ${result.details.elapsed}...`); - const preview = result.content.find(part => part.type === "text")?.text ?? ""; + const preview = replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""); return new Text(preview ? `${header}\n${theme.fg("dim", preview)}` : header, 0, 0); } const details = result.details; if (!details || !isRunDetails(details)) { - return new Text(result.content.find(part => part.type === "text")?.text ?? "", 0, 0); + return new Text(replaceTabs(result.content.find(part => part.type === "text")?.text ?? ""), 0, 0); } const statusText = renderStatus(details, theme); @@ -241,54 +363,60 @@ export function createRunExperimentTool( return new Text(statusText, 0, 0); } - const preview = options.expanded ? details.tailOutput : details.tailOutput.split("\n").slice(-5).join("\n"); + const preview = replaceTabs( + options.expanded ? details.tailOutput : details.tailOutput.split("\n").slice(-5).join("\n"), + ); const suffix = options.expanded && details.truncation && details.fullOutputPath - ? `\n${theme.fg("warning", `Full output: ${details.fullOutputPath}`)}` + ? `\n${theme.fg("warning", `Full output: ${shortenPath(details.fullOutputPath)}`)}` : ""; return new Text(preview ? `${statusText}\n${theme.fg("dim", preview)}${suffix}` : statusText, 0, 0); }, }; } -interface ProgressSnapshot { - elapsed: string; - fullOutputPath?: string; - tailOutput: string; - truncation?: RunExperimentProgressDetails["truncation"]; -} - async function executeProcess(options: { - command: string; + command: string[]; cwd: string; + logPath: string; timeoutMs: number; signal?: AbortSignal; - onProgress(details: ProgressSnapshot): void; + onProgress?(details: ProgressSnapshot): void; }): Promise { const { promise, resolve, reject } = Promise.withResolvers(); - const child = childProcess.spawn("bash", ["-lc", options.command], { + const child = childProcess.spawn(options.command[0] ?? "bash", options.command.slice(1), { cwd: options.cwd, detached: true, stdio: ["ignore", "pipe", "pipe"], }); - const getTempFile = createTempFileAllocator(); - const chunks: Buffer[] = []; + const tailChunks: Buffer[] = []; let chunksBytes = 0; - let totalBytes = 0; let killedByTimeout = false; let resolved = false; - let fullOutputPath: string | undefined; - let writeStream: fs.WriteStream | undefined; + let writeStream: fs.WriteStream | undefined = fs.createWriteStream(options.logPath); + let forceKillTimeout: NodeJS.Timeout | undefined; + + const closeWriteStream = (): Promise => { + if (!writeStream) return Promise.resolve(); + const stream = writeStream; + writeStream = undefined; + return new Promise((resolveClose, rejectClose) => { + stream.end((error?: Error | null) => { + if (error) { + rejectClose(error); + return; + } + resolveClose(); + }); + }); + }; const cleanup = (): void => { if (progressTimer) clearInterval(progressTimer); if (timeoutHandle) clearTimeout(timeoutHandle); + if (forceKillTimeout) clearTimeout(forceKillTimeout); options.signal?.removeEventListener("abort", abortHandler); - if (writeStream) { - writeStream.end(); - writeStream = undefined; - } }; const finish = (callback: () => void): void => { @@ -299,50 +427,54 @@ async function executeProcess(options: { }; const appendChunk = (data: Buffer): void => { - totalBytes += data.length; - if (!fullOutputPath && totalBytes > DEFAULT_MAX_BYTES) { - fullOutputPath = getTempFile(); - writeStream = fs.createWriteStream(fullOutputPath); - for (const chunk of chunks) { - writeStream.write(chunk); - } - } writeStream?.write(data); - chunks.push(data); + tailChunks.push(data); chunksBytes += data.length; - while (chunksBytes > DEFAULT_MAX_BYTES * 2 && chunks.length > 1) { - const removed = chunks.shift(); + while (chunksBytes > DEFAULT_MAX_BYTES * 2 && tailChunks.length > 1) { + const removed = tailChunks.shift(); if (removed) chunksBytes -= removed.length; } }; const snapshot = (): ProgressSnapshot => { - const tail = truncateTail(Buffer.concat(chunks).toString("utf8"), { + const tail = truncateTail(Buffer.concat(tailChunks).toString("utf8"), { maxBytes: DEFAULT_MAX_BYTES, maxLines: DEFAULT_MAX_LINES, }); return { elapsed: formatElapsed(Date.now() - startedAt), - fullOutputPath, + runDirectory: path.dirname(options.logPath), + fullOutputPath: options.logPath, tailOutput: tail.content, truncation: tail.truncated ? tail : undefined, }; }; + const killTreeWithEscalation = (): void => { + if (!child.pid) return; + killTree(child.pid); + forceKillTimeout = setTimeout(() => { + if (child.pid) killTree(child.pid, "SIGKILL"); + }, 1_000); + forceKillTimeout.unref?.(); + }; + const startedAt = Date.now(); - const progressTimer = setInterval(() => { - options.onProgress(snapshot()); - }, 1000); + const progressTimer = options.onProgress + ? setInterval(() => { + options.onProgress?.(snapshot()); + }, 1000) + : undefined; const timeoutHandle = options.timeoutMs > 0 ? setTimeout(() => { killedByTimeout = true; - if (child.pid) killTree(child.pid); + killTreeWithEscalation(); }, options.timeoutMs) : undefined; const abortHandler = (): void => { - if (child.pid) killTree(child.pid); + killTreeWithEscalation(); }; if (options.signal?.aborted) { abortHandler(); @@ -357,50 +489,59 @@ async function executeProcess(options: { appendChunk(data); }); child.on("error", error => { - finish(() => reject(error)); + void closeWriteStream().finally(() => { + finish(() => reject(error)); + }); }); - child.on("close", code => { - if (options.signal?.aborted) { - finish(() => reject(new Error("aborted"))); - return; + child.on("close", async code => { + try { + await closeWriteStream(); + if (options.signal?.aborted) { + finish(() => reject(new Error("aborted"))); + return; + } + const output = await fs.promises.readFile(options.logPath, "utf8"); + finish(() => + resolve({ + exitCode: code, + killed: killedByTimeout, + logPath: options.logPath, + output, + }), + ); + } catch (error) { + finish(() => reject(error)); } - const output = Buffer.concat(chunks).toString("utf8"); - finish(() => - resolve({ - actualTotalBytes: totalBytes, - exitCode: code, - killed: killedByTimeout, - output, - tempFilePath: fullOutputPath, - }), - ); }); return promise; } -function runChecks(options: { +async function runChecks(options: { cwd: string; pathToChecks: string; + logPath: string; timeoutMs: number; signal?: AbortSignal; - // signal currently unused because spawnSync does not support AbortSignal directly. -}): ChecksExecutionResult { - const result = childProcess.spawnSync("bash", [options.pathToChecks], { +}): Promise { + const result = await executeProcess({ + command: ["bash", options.pathToChecks], cwd: options.cwd, - timeout: options.timeoutMs, - encoding: "utf8", - maxBuffer: DEFAULT_MAX_BYTES, + logPath: options.logPath, + timeoutMs: options.timeoutMs, + signal: options.signal, }); return { - code: result.status, - killed: result.signal === "SIGTERM" || result.signal === "SIGKILL" || Boolean(result.error), - output: `${result.stdout ?? ""}${result.stderr ?? ""}`.trim(), + code: result.exitCode, + killed: result.killed, + logPath: result.logPath, + output: result.output.trim(), }; } function buildRunText(details: RunDetails, outputPreview: string, bestMetric: number | null): string { const lines: string[] = []; + lines.push(`Run directory: ${details.runDirectory}`); if (details.timedOut) { lines.push(`TIMEOUT after ${details.durationSeconds.toFixed(1)}s`); } else if (details.exitCode !== 0) { @@ -420,13 +561,16 @@ function buildRunText(details: RunDetails, outputPreview: string, bestMetric: nu } if (details.parsedPrimary !== null) { lines.push(`Parsed ${details.metricName}: ${details.parsedPrimary}`); + lines.push(`Next log_experiment metric: ${details.parsedPrimary}`); } if (details.parsedMetrics) { - const secondary = Object.entries(details.parsedMetrics) + const secondaryEntries = Object.entries(details.parsedMetrics) .filter(([name]) => name !== details.metricName) - .map(([name, value]) => `${name}=${value}`); + .map(([name, value]) => [name, value] as const); + const secondary = secondaryEntries.map(([name, value]) => `${name}=${value}`); if (secondary.length > 0) { lines.push(`Parsed metrics: ${secondary.join(", ")}`); + lines.push(`Next log_experiment metrics: ${JSON.stringify(Object.fromEntries(secondaryEntries))}`); } } if (details.parsedAsi) { @@ -440,6 +584,9 @@ function buildRunText(details: RunDetails, outputPreview: string, bestMetric: nu `Output truncated (${formatBytes(EXPERIMENT_MAX_BYTES)} limit). Full output: ${details.fullOutputPath}`, ); } + if (details.checksLogPath) { + lines.push(`Checks log: ${details.checksLogPath}`); + } if (details.checksPass === false && details.checksOutput.length > 0) { lines.push(""); lines.push("Checks output:"); @@ -477,3 +624,13 @@ function isProgressDetails(value: unknown): value is RunExperimentProgressDetail if (typeof value !== "object" || value === null) return false; return "phase" in value && value.phase === "running"; } + +function collectLoggedRunNumbers(results: Array<{ runNumber: number | null }>): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} diff --git a/packages/coding-agent/src/autoresearch/types.ts b/packages/coding-agent/src/autoresearch/types.ts index 52f7b89ec..27ec82d02 100644 --- a/packages/coding-agent/src/autoresearch/types.ts +++ b/packages/coding-agent/src/autoresearch/types.ts @@ -21,7 +21,23 @@ export interface MetricDef { unit: string; } +export interface AutoresearchBenchmarkContract { + command: string | null; + primaryMetric: string | null; + metricUnit: string; + direction: MetricDirection | null; + secondaryMetrics: string[]; +} + +export interface AutoresearchContract { + benchmark: AutoresearchBenchmarkContract; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; +} + export interface ExperimentResult { + runNumber: number | null; commit: string; metric: number; metrics: NumericMetricMap; @@ -44,6 +60,11 @@ export interface ExperimentState { currentSegment: number; maxExperiments: number | null; confidence: number | null; + benchmarkCommand: string | null; + scopePaths: string[]; + offLimits: string[]; + constraints: string[]; + segmentFingerprint: string | null; } export interface RunExperimentProgressDetails { @@ -51,9 +72,14 @@ export interface RunExperimentProgressDetails { elapsed: string; truncation?: TruncationResult; fullOutputPath?: string; + runDirectory?: string; } export interface RunDetails { + runNumber: number; + runDirectory: string; + benchmarkLogPath: string; + checksLogPath?: string; command: string; exitCode: number | null; durationSeconds: number; @@ -86,20 +112,36 @@ export interface ChecksResult { duration: number; } +export interface PendingRunSummary { + checksDurationSeconds: number | null; + checksPass: boolean | null; + checksTimedOut: boolean; + command: string; + durationSeconds: number | null; + parsedAsi: ASIData | null; + parsedMetrics: NumericMetricMap | null; + parsedPrimary: number | null; + passed: boolean; + runDirectory: string; + runNumber: number; +} + export interface RunningExperiment { startedAt: number; command: string; + runDirectory: string; + runNumber: number; } export interface AutoresearchRuntime { autoresearchMode: boolean; dashboardExpanded: boolean; - lastAutoResumeTime: number; - experimentsThisSession: number; - autoResumeTurns: number; lastRunChecks: ChecksResult | null; lastRunDuration: number | null; lastRunAsi: ASIData | null; + lastRunArtifactDir: string | null; + lastRunNumber: number | null; + lastRunSummary: PendingRunSummary | null; runningExperiment: RunningExperiment | null; state: ExperimentState; goal: string | null; @@ -116,6 +158,12 @@ export interface AutoresearchJsonConfigEntry { metricName?: string; metricUnit?: string; bestDirection?: MetricDirection; + benchmarkCommand?: string; + secondaryMetrics?: string[]; + scopePaths?: string[]; + offLimits?: string[]; + constraints?: string[]; + segmentFingerprint?: string; } export interface AutoresearchJsonRunEntry { @@ -143,6 +191,7 @@ export interface AutoresearchControlEntryData { export interface ReconstructedControlState { autoresearchMode: boolean; goal: string | null; + lastMode: AutoresearchControlEntryData["mode"] | null; } export interface RuntimeStore { diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 6f556a6dd..c61fd3d9b 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -1054,7 +1054,13 @@ export interface ExtensionAPI { // Actions // ========================================================================= - /** Send a custom message to the session. */ + /** + * Send a custom message to the session. + * + * `deliverAs: "nextTurn"` keeps the message hidden from the editable pending-message UI. + * If `triggerTurn` is also true while the current turn is still unwinding, the session schedules + * an internal continuation that consumes the message on the next turn. + */ sendMessage( message: Pick, "customType" | "content" | "display" | "details" | "attribution">, options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" }, @@ -1230,6 +1236,11 @@ type HandlerFn = (...args: unknown[]) => Promise; export type SendMessageHandler = ( message: Pick, "customType" | "content" | "display" | "details" | "attribution">, + /** + * `deliverAs: "nextTurn"` queues hidden custom context for the next turn. + * When paired with `triggerTurn: true` during prompt teardown, the session schedules + * an internal continuation without surfacing the message in the editable pending queue. + */ options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" }, ) => void; diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 108c777c7..ffe123444 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -364,6 +364,7 @@ export class AgentSession { #followUpMessages: string[] = []; /** Messages queued to be included with the next user prompt as context ("asides"). */ #pendingNextTurnMessages: CustomMessage[] = []; + #scheduledHiddenNextTurnGeneration: number | undefined = undefined; #planModeState: PlanModeState | undefined; #planReferenceSent = false; #planReferencePath = "local://PLAN.md"; @@ -2567,6 +2568,74 @@ export class AgentSession { }); } + #queueHiddenNextTurnMessage(message: CustomMessage, triggerTurn: boolean): void { + this.#pendingNextTurnMessages.push(message); + if (!triggerTurn) return; + const generation = this.#promptGeneration; + if (this.#scheduledHiddenNextTurnGeneration === generation) { + return; + } + this.#scheduledHiddenNextTurnGeneration = generation; + this.#schedulePostPromptTask( + async () => { + if (this.#scheduledHiddenNextTurnGeneration === generation) { + this.#scheduledHiddenNextTurnGeneration = undefined; + } + if (this.#pendingNextTurnMessages.length === 0) { + return; + } + try { + await this.#promptQueuedHiddenNextTurnMessages(); + } catch { + // Leave the hidden next-turn messages queued for the next explicit prompt. + } + }, + { + generation, + onSkip: () => { + if (this.#scheduledHiddenNextTurnGeneration === generation) { + this.#scheduledHiddenNextTurnGeneration = undefined; + } + }, + }, + ); + } + + async #promptQueuedHiddenNextTurnMessages(): Promise { + if (this.#pendingNextTurnMessages.length === 0) { + return; + } + + const queuedMessages = [...this.#pendingNextTurnMessages]; + this.#pendingNextTurnMessages = []; + const message = queuedMessages[queuedMessages.length - 1]; + if (!message) { + return; + } + + const prependMessages = queuedMessages.slice(0, -1); + const textContent = this.#getCustomMessageTextContent(message); + try { + await this.#promptWithMessage(message, textContent, { + prependMessages, + skipPostPromptRecoveryWait: true, + }); + } catch (error) { + this.#pendingNextTurnMessages = [...queuedMessages, ...this.#pendingNextTurnMessages]; + throw error; + } + } + + #getCustomMessageTextContent(message: Pick): string { + if (typeof message.content === "string") { + return message.content; + } + return message.content + .filter((content): content is TextContent => content.type === "text") + .map(content => content.text) + .join(""); + } + /** * Throw an error if the text is an extension command. */ @@ -2607,7 +2676,7 @@ export class AgentSession { }; if (this.isStreaming) { if (options?.deliverAs === "nextTurn") { - this.#pendingNextTurnMessages.push(appMessage); + this.#queueHiddenNextTurnMessage(appMessage, options?.triggerTurn ?? false); return; } @@ -2619,6 +2688,22 @@ export class AgentSession { return; } + if (options?.deliverAs === "nextTurn") { + if (options?.triggerTurn) { + await this.agent.prompt(appMessage); + return; + } + this.agent.appendMessage(appMessage); + this.sessionManager.appendCustomMessageEntry( + message.customType, + message.content, + message.display, + message.details, + message.attribution ?? "agent", + ); + return; + } + if (options?.triggerTurn) { await this.agent.prompt(appMessage); return; @@ -2686,9 +2771,9 @@ export class AgentSession { return { steering, followUp }; } - /** Number of pending messages (includes both steering and follow-up) */ + /** Number of pending messages (includes steering, follow-up, and next-turn messages) */ get queuedMessageCount(): number { - return this.#steeringMessages.length + this.#followUpMessages.length; + return this.#steeringMessages.length + this.#followUpMessages.length + this.#pendingNextTurnMessages.length; } /** Get pending messages (read-only) */ @@ -2830,6 +2915,7 @@ export class AgentSession { async abort(): Promise { this.abortRetry(); this.#promptGeneration++; + this.#scheduledHiddenNextTurnGeneration = undefined; this.#resolveTtsrResume(); this.#cancelPostPromptTasks(); this.agent.abort(); @@ -2879,6 +2965,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; this.sessionManager.appendThinkingLevelChange(this.thinkingLevel); this.sessionManager.appendServiceTierChange(this.serviceTier ?? null); @@ -3612,6 +3699,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; this.#todoReminderCount = 0; // Inject the handoff document as a custom message @@ -4961,6 +5049,7 @@ export class AgentSession { this.#steeringMessages = []; this.#followUpMessages = []; this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; // Flush pending writes before switching await this.sessionManager.flush(); @@ -5060,6 +5149,7 @@ export class AgentSession { // Clear pending messages (bound to old session state) this.#pendingNextTurnMessages = []; + this.#scheduledHiddenNextTurnGeneration = undefined; // Flush pending writes before branching await this.sessionManager.flush(); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 88408f1d6..42ccc162a 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -2,11 +2,11 @@ * Tests for AgentSession concurrent prompt guard. */ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { Agent, AgentBusyError, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; import { type AssistantMessage, getBundledModel, type ToolCall } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; @@ -62,6 +62,7 @@ describe("AgentSession concurrent prompt guard", () => { if (tempDir && fs.existsSync(tempDir)) { fs.rmSync(tempDir, { recursive: true }); } + vi.restoreAllMocks(); }); async function createSession() { @@ -163,6 +164,76 @@ describe("AgentSession concurrent prompt guard", () => { await firstPrompt.catch(() => {}); }); + it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let firstStream: MockAssistantStream | undefined; + const callMessages: AgentMessage[][] = []; + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { + model, + systemPrompt: "Test", + tools: [], + }, + streamFn: (_model, context) => { + callMessages.push([...context.messages]); + const stream = new MockAssistantStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: createAssistantMessage("") }); + if (callMessages.length > 1) { + stream.push({ type: "done", reason: "stop", message: createAssistantMessage("Resumed") }); + return; + } + }); + firstStream = stream; + return stream; + }, + }); + + const sessionManager = SessionManager.inMemory(); + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + + session = new AgentSession({ + agent, + sessionManager, + settings, + modelRegistry, + }); + + const firstPrompt = session.prompt("First message"); + await Bun.sleep(10); + + await session.sendCustomMessage( + { + customType: "autoresearch-resume", + content: "Hidden stop reaction", + display: false, + attribution: "agent", + }, + { deliverAs: "nextTurn", triggerTurn: true }, + ); + + expect(session.queuedMessageCount).toBe(0); + expect(session.getQueuedMessages()).toEqual({ steering: [], followUp: [] }); + + firstStream?.push({ type: "done", reason: "stop", message: createAssistantMessage("Done") }); + await firstPrompt; + await session.waitForIdle(); + + expect(callMessages).toHaveLength(2); + expect( + callMessages[1]?.some( + message => + message.role === "custom" && "customType" in message && message.customType === "autoresearch-resume", + ), + ).toBe(true); + }); + it("should allow prompt() after previous completes", async () => { // Create session with a stream that completes immediately const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; diff --git a/packages/coding-agent/test/autoresearch-state.test.ts b/packages/coding-agent/test/autoresearch-state.test.ts index 285cd70da..a5d20eb17 100644 --- a/packages/coding-agent/test/autoresearch-state.test.ts +++ b/packages/coding-agent/test/autoresearch-state.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Snowflake } from "@oh-my-pi/pi-utils"; +import { parseAutoresearchContract } from "../src/autoresearch/contract"; import { isAutoresearchShCommand } from "../src/autoresearch/helpers"; import { createAutoresearchExtension } from "../src/autoresearch/index"; import { reconstructStateFromJsonl } from "../src/autoresearch/state"; @@ -14,6 +15,7 @@ import type { RegisteredCommand, SessionStartEvent, SessionSwitchEvent, + ToolCallEvent, } from "../src/extensibility/extensions"; function makeTempDir(): string { @@ -100,6 +102,187 @@ describe("autoresearch state reconstruction", () => { expect(state.results.filter(result => result.segment === 1)).toHaveLength(2); expect(state.secondaryMetrics).toEqual([{ name: "latency_ms", unit: "ms" }]); }); + + it("hydrates configured secondary metrics from config entries before later runs add new ones", () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const jsonlPath = path.join(dir, "autoresearch.jsonl"); + fs.writeFileSync( + jsonlPath, + [ + JSON.stringify({ + type: "config", + name: "Baseline", + metricName: "runtime_ms", + metricUnit: "ms", + bestDirection: "lower", + secondaryMetrics: ["memory_mb", "tokens"], + }), + JSON.stringify({ + commit: "aaaaaaa", + metric: 100, + metrics: { memory_mb: 32 }, + status: "keep", + description: "baseline", + timestamp: 1, + }), + ].join("\n"), + ); + + const reconstructed = reconstructStateFromJsonl(dir); + expect(reconstructed.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + }); + + it("uses the first kept run as baseline and preserves configured secondary metrics before they appear", () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const jsonlPath = path.join(dir, "autoresearch.jsonl"); + fs.writeFileSync( + jsonlPath, + [ + JSON.stringify({ + type: "config", + name: "Baseline after crash", + metricName: "runtime_ms", + metricUnit: "ms", + bestDirection: "lower", + secondaryMetrics: ["memory_mb", "tokens"], + }), + JSON.stringify({ + commit: "aaaaaaa", + metric: 0, + status: "crash", + description: "broken first run", + timestamp: 1, + }), + JSON.stringify({ + commit: "bbbbbbb", + metric: 120, + metrics: { memory_mb: 32 }, + status: "keep", + description: "baseline", + timestamp: 2, + }), + ].join("\n"), + ); + + const reconstructed = reconstructStateFromJsonl(dir); + expect(reconstructed.state.bestMetric).toBe(120); + expect(reconstructed.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + }); + + it("parses benchmark, scope, off-limits, and constraints from autoresearch.md", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower +- secondary metrics: memory_mb, tokens + +## Files in Scope +- src/core +- src/feature.ts + +## Off Limits +- src/generated + +## Constraints +- keep API stable +- no behavior regressions +`); + + expect(contract.benchmark.command).toBe("bash autoresearch.sh"); + expect(contract.benchmark.primaryMetric).toBe("runtime_ms"); + expect(contract.benchmark.metricUnit).toBe("ms"); + expect(contract.benchmark.direction).toBe("lower"); + expect(contract.benchmark.secondaryMetrics).toEqual(["memory_mb", "tokens"]); + expect(contract.scopePaths).toEqual(["src/core", "src/feature.ts"]); + expect(contract.offLimits).toEqual(["src/generated"]); + expect(contract.constraints).toEqual(["keep API stable", "no behavior regressions"]); + }); + + it("parses nested secondary metric bullets from autoresearch.md", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower +- secondary metrics: + - memory_mb + - rss_mb + +## Files in Scope +- src +`); + + expect(contract.benchmark.secondaryMetrics).toEqual(["memory_mb", "rss_mb"]); + }); + + it("allows empty optional sections while preserving an empty off-limits list", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: +- direction: higher + +## Files in Scope +- . + +## Off Limits + +## Constraints +`); + + expect(contract.benchmark.metricUnit).toBe(""); + expect(contract.benchmark.direction).toBe("higher"); + expect(contract.scopePaths).toEqual(["."]); + expect(contract.offLimits).toEqual([]); + expect(contract.constraints).toEqual([]); + }); + + it("preserves free-form constraint text without path normalization", () => { + const contract = parseAutoresearchContract(` +# Autoresearch + +## Benchmark +- command: bash autoresearch.sh +- primary metric: runtime_ms +- metric unit: ms +- direction: lower + +## Files in Scope +- src/ + +## Off Limits +- generated/ + +## Constraints +- keep docs/ wording exactly as written +- do not rewrite ./README.md examples +`); + + expect(contract.scopePaths).toEqual(["src"]); + expect(contract.offLimits).toEqual(["generated"]); + expect(contract.constraints).toEqual([ + "keep docs/ wording exactly as written", + "do not rewrite ./README.md examples", + ]); + }); }); describe("autoresearch command guard", () => { @@ -127,7 +310,7 @@ interface AutoresearchCommandHarness { function createAutoresearchCommandHarness( cwd: string, - inputResult: string | undefined, + inputResult: string | string[] | undefined, execImpl?: (command: string, args: string[]) => Promise<{ code: number; stderr: string; stdout: string }>, ): AutoresearchCommandHarness { const execCalls: Array<{ args: string[]; command: string }> = []; @@ -135,6 +318,7 @@ function createAutoresearchCommandHarness( const inputCalls: Array<{ title: string; placeholder: string | undefined }> = []; const notifications: Array<{ message: string; type: "info" | "warning" | "error" | undefined }> = []; let command: RegisteredCommand | undefined; + const inputQueue = typeof inputResult === "string" || inputResult === undefined ? [inputResult] : [...inputResult]; const api = { appendEntry(_customType: string, _data?: unknown): void {}, @@ -178,6 +362,7 @@ function createAutoresearchCommandHarness( newSession: async () => ({ cancelled: false }), reload: async () => {}, sessionManager: { + getBranch: () => [], getEntries: () => [], getSessionId: () => "session-1", }, @@ -188,7 +373,7 @@ function createAutoresearchCommandHarness( custom: async () => undefined, input: async (title: string, placeholder?: string) => { inputCalls.push({ title, placeholder }); - return inputResult; + return inputQueue.shift(); }, notify(message: string, type?: "info" | "warning" | "error"): void { notifications.push({ message, type }); @@ -211,17 +396,23 @@ function createAutoresearchCommandHarness( interface AutoresearchLifecycleHarness { sessionStartHandler: ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) | undefined; sessionSwitchHandler: ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined; + agentEndHandler: ((event: unknown, ctx: ExtensionContext) => Promise | void) | undefined; + toolCallHandler: ((event: ToolCallEvent, ctx: ExtensionContext) => Promise | unknown) | undefined; ctx: ExtensionContext; setActiveToolsCalls: string[][]; + sentMessages: Array<{ message: unknown; options: unknown }>; } function createAutoresearchLifecycleHarness(options: { activeTools: string[]; + branchEntries?: Array<{ type: "custom"; customType: string; data?: unknown }>; controlEntries?: Array<{ type: "custom"; customType: string; data?: unknown }>; + cwd?: string; }): AutoresearchLifecycleHarness { const handlers = new Map Promise | void>(); const activeTools = [...options.activeTools]; const setActiveToolsCalls: string[][] = []; + const sentMessages: Array<{ message: unknown; options: unknown }> = []; const api = { appendEntry(_customType: string, _data?: unknown): void {}, @@ -234,6 +425,9 @@ function createAutoresearchLifecycleHarness(options: { getActiveTools(): string[] { return [...activeTools]; }, + sendMessage(message: unknown, options?: unknown): void { + sentMessages.push({ message, options }); + }, async setActiveTools(toolNames: string[]): Promise { setActiveToolsCalls.push([...toolNames]); activeTools.splice(0, activeTools.length, ...toolNames); @@ -245,7 +439,7 @@ function createAutoresearchLifecycleHarness(options: { const ctx = { abort(): void {}, compact: async () => {}, - cwd: makeTempDir(), + cwd: options.cwd ?? makeTempDir(), getContextUsage: () => undefined, hasUI: false, hasPendingMessages: () => false, @@ -253,6 +447,7 @@ function createAutoresearchLifecycleHarness(options: { model: undefined, modelRegistry: {}, sessionManager: { + getBranch: () => options.branchEntries ?? options.controlEntries ?? [], getEntries: () => options.controlEntries ?? [], getSessionId: () => "session-1", }, @@ -286,8 +481,15 @@ function createAutoresearchLifecycleHarness(options: { sessionSwitchHandler: handlers.get("session_switch") as | ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined, + agentEndHandler: handlers.get("agent_end") as + | ((event: unknown, ctx: ExtensionContext) => Promise | void) + | undefined, + toolCallHandler: handlers.get("tool_call") as + | ((event: ToolCallEvent, ctx: ExtensionContext) => Promise | unknown) + | undefined, ctx, setActiveToolsCalls, + sentMessages, }; } @@ -307,7 +509,16 @@ describe("autoresearch command startup", () => { const branches = new Set(); const harness = createAutoresearchCommandHarness( dir, - "reduce edit benchmark runtime variance", + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh --quick", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch, packages/coding-agent/test", + "packages/coding-agent/src/generated", + "preserve output format", + ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; @@ -332,13 +543,27 @@ describe("autoresearch command startup", () => { expect(harness.inputCalls).toEqual([ { title: "Autoresearch Intent", placeholder: "what should autoresearch improve?" }, + { title: "Benchmark Command", placeholder: "bash autoresearch.sh" }, + { title: "Primary Metric Name", placeholder: "runtime_ms" }, + { title: "Metric Unit", placeholder: "ms" }, + { title: "Metric Direction", placeholder: "lower" }, + { title: "Files in Scope", placeholder: "packages/coding-agent/src/autoresearch" }, + { title: "Off Limits", placeholder: "" }, + { title: "Constraints", placeholder: "" }, ]); expect(harness.sentMessages).toHaveLength(1); expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); expect(harness.sentMessages[0]).toContain("reduce edit benchmark runtime variance"); + expect(harness.sentMessages[0]).toContain("benchmark command: `bash autoresearch.sh --quick`"); + expect(harness.sentMessages[0]).toContain("primary metric: `runtime_ms`"); + expect(harness.sentMessages[0]).toContain("metric unit: `ms`"); + expect(harness.sentMessages[0]).toContain("direction: `lower`"); + expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/autoresearch`"); + expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/generated`"); + expect(harness.sentMessages[0]).toContain("preserve output format"); expect(harness.sentMessages[0]).toContain("Created and checked out dedicated git branch"); expect(harness.sentMessages[0]).toContain("Explain briefly what autoresearch will do in this repository"); - expect(harness.sentMessages[0]).toContain("Files in Scope"); + expect(harness.sentMessages[0]).toContain("- files in scope:"); expect(harness.notifications).toEqual([]); const checkoutCall = harness.execCalls.find(call => call.command === "git" && call.args[0] === "checkout"); expect(checkoutCall?.args[2]).toMatch(/^autoresearch\/reduce-edit-benchmark-runtime-variance-\d{8}$/); @@ -352,6 +577,7 @@ describe("autoresearch command startup", () => { const harness = createAutoresearchCommandHarness(dir, "ignored", async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; if (args[0] === "branch" && args[1] === "--show-current") { return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; } @@ -378,6 +604,108 @@ describe("autoresearch command startup", () => { ]); }); + it("includes explicit resume context when the user resumes with additional instructions", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const autoresearchMdPath = path.join(dir, "autoresearch.md"); + fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); + await Bun.write(path.join(dir, ".autoresearch", "runs", "0001", "run.json"), "{}"); + const harness = createAutoresearchCommandHarness(dir, undefined, async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }); + + await harness.command.handler("focus on memory regressions next", harness.ctx); + + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Additional context from the user:"); + expect(harness.sentMessages[0]).toContain("focus on memory regressions next"); + expect(harness.sentMessages[0]).toContain(`@${autoresearchMdPath}`); + }); + + it("treats an explicit new intent as a fresh setup when only stale notes remain", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync(path.join(dir, "autoresearch.md"), "# Autoresearch\n\nOld notes\n"); + let currentBranch = "main"; + const branches = new Set(); + const harness = createAutoresearchCommandHarness( + dir, + [ + "focus on memory regressions next", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: `${currentBranch}\n` }; + } + if (args[0] === "status") return { code: 0, stderr: "", stdout: "" }; + if (args[0] === "show-ref") { + const branchName = args[args.length - 1]?.replace("refs/heads/", "") ?? ""; + return { code: branches.has(branchName) ? 0 : 1, stderr: "", stdout: "" }; + } + if (args[0] === "checkout" && args[1] === "-b") { + currentBranch = args[2] ?? currentBranch; + branches.add(currentBranch); + return { code: 0, stderr: "", stdout: "" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await harness.command.handler("focus on memory regressions next", harness.ctx); + + expect(harness.inputCalls[0]).toEqual({ + title: "Autoresearch Intent", + placeholder: "focus on memory regressions next", + }); + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Set up autoresearch for this intent:"); + expect(harness.sentMessages[0]).not.toContain("Resume autoresearch from the attached notes."); + }); + + it("refuses to resume on an autoresearch branch when non-local files are dirty", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const autoresearchMdPath = path.join(dir, "autoresearch.md"); + fs.writeFileSync(autoresearchMdPath, "# Autoresearch\n\nExisting notes\n"); + const harness = createAutoresearchCommandHarness(dir, "ignored", async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: " M packages/coding-agent/src/sdk.ts\0" }; + } + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "autoresearch/existing-20260322\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([ + { + message: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", + type: "error", + }, + ]); + }); + it("does not start autoresearch when the intent dialog returns blank input", async () => { const dir = makeTempDir(); tempDirs.push(dir); @@ -389,12 +717,34 @@ describe("autoresearch command startup", () => { expect(harness.notifications).toEqual([{ message: "Autoresearch intent is required", type: "info" }]); }); + it("rejects non-canonical benchmark commands during setup", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + const harness = createAutoresearchCommandHarness(dir, ["speed things up", "pnpm test"]); + + await harness.command.handler("", harness.ctx); + + expect(harness.sentMessages).toEqual([]); + expect(harness.notifications).toEqual([ + { message: "Benchmark command must invoke `autoresearch.sh` directly", type: "info" }, + ]); + }); + it("refuses to start when non-autoresearch files are dirty on a non-autoresearch branch", async () => { const dir = makeTempDir(); tempDirs.push(dir); const harness = createAutoresearchCommandHarness( dir, - "reduce edit benchmark runtime variance", + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; @@ -414,11 +764,299 @@ describe("autoresearch command startup", () => { expect(harness.notifications).toEqual([ { message: - "Autoresearch needs a clean git worktree before it can create an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: packages/coding-agent/src/sdk.ts", type: "error", }, ]); }); + + it("ignores autoresearch local state but still blocks dirty control files before creating a branch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + + const localStateHarness = createAutoresearchCommandHarness( + dir, + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "main\n" }; + } + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: "?? autoresearch.jsonl\n?? .autoresearch/runs/0001/run.json\n" }; + } + if (args[0] === "show-ref") return { code: 1, stderr: "", stdout: "" }; + if (args[0] === "checkout" && args[1] === "-b") return { code: 0, stderr: "", stdout: "" }; + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await localStateHarness.command.handler("", localStateHarness.ctx); + + expect(localStateHarness.sentMessages).toHaveLength(1); + expect(localStateHarness.notifications).toEqual([]); + + const dirtyControlHarness = createAutoresearchCommandHarness( + dir, + [ + "reduce edit benchmark runtime variance", + "bash autoresearch.sh", + "runtime_ms", + "ms", + "lower", + "packages/coding-agent/src/autoresearch", + "", + "", + ], + async (command, args) => { + if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; + if (args[0] === "branch" && args[1] === "--show-current") { + return { code: 0, stderr: "", stdout: "main\n" }; + } + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: " M autoresearch.md\n" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + ); + + await dirtyControlHarness.command.handler("", dirtyControlHarness.ctx); + + expect(dirtyControlHarness.sentMessages).toEqual([]); + expect(dirtyControlHarness.notifications).toEqual([ + { + message: + "Autoresearch needs a clean git worktree before it can create or reuse an isolated branch. Commit or stash these paths first: autoresearch.md", + type: "error", + }, + ]); + }); +}); + +describe("autoresearch tool-call guard", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("blocks out-of-scope edits but allows autoresearch control files", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ + type: "config", + metricName: "runtime_ms", + metricUnit: "ms", + scopePaths: ["src"], + offLimits: ["src/generated"], + })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blockedScope = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-1", + toolName: "write", + input: { path: "README.md", content: "nope" }, + }, + harness.ctx, + ); + expect(blockedScope).toEqual({ + block: true, + reason: expect.stringContaining("outside Files in Scope"), + }); + + const blockedLocalState = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-2", + toolName: "write", + input: { path: "autoresearch.jsonl", content: "[]" }, + }, + harness.ctx, + ); + expect(blockedLocalState).toEqual({ + block: true, + reason: expect.stringContaining("local state files"), + }); + + const allowedControl = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-3", + toolName: "write", + input: { path: "autoresearch.program.md", content: "# Strategy" }, + }, + harness.ctx, + ); + expect(allowedControl).toBeUndefined(); + }); + + it("requires ast_edit to declare an explicit path during autoresearch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-ast", + toolName: "ast_edit", + input: { ops: [{ pat: "a", out: "b" }] }, + }, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("explicit target path"), + }); + }); + + it("blocks mutating bash commands during autoresearch", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-bash", + toolName: "bash", + input: { command: "rm -rf src/generated" }, + } as ToolCallEvent, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("read-only shell inspection"), + }); + }); + + it("blocks symlink escapes that point outside the working tree", async () => { + const dir = makeTempDir(); + const outsideDir = makeTempDir(); + tempDirs.push(dir, outsideDir); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.symlinkSync(outsideDir, path.join(dir, "src", "linked-outside"), "dir"); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: [], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + + const blocked = await harness.toolCallHandler?.( + { + type: "tool_call", + toolCallId: "call-symlink", + toolName: "write", + input: { path: "src/linked-outside/escape.ts", content: "export const value = 1;\n" }, + }, + harness.ctx, + ); + expect(blocked).toEqual({ + block: true, + reason: expect.stringContaining("outside the working tree"), + }); + }); +}); + +describe("autoresearch auto-resume", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("includes the pending-run reminder after rehydrate when agent_end schedules an auto-resume", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", metricName: "runtime_ms", scopePaths: ["src"] })}\n`, + ); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["init_experiment", "run_experiment", "log_experiment"], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + await harness.agentEndHandler?.({}, harness.ctx); + + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]?.message).toMatchObject({ + customType: "autoresearch-resume", + content: expect.stringContaining("finish the pending `log_experiment` step"), + }); + expect(harness.sentMessages[0]?.options).toMatchObject({ + deliverAs: "nextTurn", + triggerTurn: true, + }); + }); }); describe("autoresearch lifecycle tool activation", () => { @@ -449,6 +1087,19 @@ describe("autoresearch lifecycle tool activation", () => { expect(harness.setActiveToolsCalls).toEqual([["read"]]); }); + + it("rehydrates control state from the active branch only", async () => { + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["read"], + branchEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "off" } }], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "speed" } }], + }); + + if (!harness.sessionStartHandler) throw new Error("Expected session_start handler"); + await harness.sessionStartHandler({ type: "session_start" }, harness.ctx); + + expect(harness.setActiveToolsCalls).toEqual([]); + }); }); describe("autoresearch ASI requirements", () => { diff --git a/packages/coding-agent/test/autoresearch-tools.test.ts b/packages/coding-agent/test/autoresearch-tools.test.ts new file mode 100644 index 000000000..068c16452 --- /dev/null +++ b/packages/coding-agent/test/autoresearch-tools.test.ts @@ -0,0 +1,1231 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { $ } from "bun"; +import { + buildAutoresearchSegmentFingerprint, + loadAutoresearchScriptSnapshot, + readAutoresearchContract, +} from "../src/autoresearch/contract"; +import { readPendingRunSummary } from "../src/autoresearch/helpers"; +import { createSessionRuntime } from "../src/autoresearch/state"; +import { createInitExperimentTool } from "../src/autoresearch/tools/init-experiment"; +import { createLogExperimentTool } from "../src/autoresearch/tools/log-experiment"; +import { createRunExperimentTool } from "../src/autoresearch/tools/run-experiment"; +import type { RunDetails } from "../src/autoresearch/types"; +import type { ExtensionAPI, ExtensionContext } from "../src/extensibility/extensions"; + +function makeTempDir(): string { + const dir = path.join(os.tmpdir(), `pi-autoresearch-tools-${Snowflake.next()}`); + fs.mkdirSync(dir, { recursive: true }); + return dir; +} + +function writeAutoresearchWorkspace( + dir: string, + options?: { + checksScript?: string; + contract?: string; + benchmarkScript?: string; + }, +): void { + fs.writeFileSync( + path.join(dir, "autoresearch.md"), + options?.contract ?? + [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + ); + fs.writeFileSync( + path.join(dir, "autoresearch.sh"), + options?.benchmarkScript ?? + [ + "#!/usr/bin/env bash", + "set -euo pipefail", + "echo METRIC runtime_ms=10", + "echo METRIC memory_mb=32", + 'echo ASI hypothesis="baseline"', + ].join("\n"), + ); + fs.chmodSync(path.join(dir, "autoresearch.sh"), 0o755); + if (options?.checksScript) { + fs.writeFileSync(path.join(dir, "autoresearch.checks.sh"), options.checksScript); + fs.chmodSync(path.join(dir, "autoresearch.checks.sh"), 0o755); + } +} + +function createFingerprint(workDir: string): string { + const contractResult = readAutoresearchContract(workDir); + const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); + if (contractResult.errors.length > 0 || scriptSnapshot.errors.length > 0) { + throw new Error(`Workspace setup invalid: ${[...contractResult.errors, ...scriptSnapshot.errors].join(" ")}`); + } + return buildAutoresearchSegmentFingerprint(contractResult.contract, { + benchmarkScript: scriptSnapshot.benchmarkScript, + checksScript: scriptSnapshot.checksScript, + }); +} + +function createDashboardStub() { + return { + clear(): void {}, + requestRender(): void {}, + showOverlay: async (): Promise => {}, + updateWidget(): void {}, + }; +} + +function createContext(cwd: string): ExtensionContext { + return { cwd, hasUI: false } as ExtensionContext; +} + +function createGitApi(): ExtensionAPI { + return { + exec: async (command: string, args: string[], options?: { cwd?: string }) => { + const result = Bun.spawnSync([command, ...args], { + cwd: options?.cwd ?? process.cwd(), + stdout: "pipe", + stderr: "pipe", + }); + return { + code: result.exitCode, + stdout: Buffer.from(result.stdout).toString("utf8"), + stderr: Buffer.from(result.stderr).toString("utf8"), + }; + }, + } as unknown as ExtensionAPI; +} + +function createManagedGitApi(options?: { activeTools?: string[] }) { + const activeTools = [...(options?.activeTools ?? ["init_experiment", "run_experiment", "log_experiment"])]; + const appendEntries: Array<{ customType: string; data: unknown }> = []; + const setActiveToolsCalls: string[][] = []; + const api = { + appendEntry: (customType: string, data?: unknown) => { + appendEntries.push({ customType, data }); + }, + exec: async (command: string, args: string[], execOptions?: { cwd?: string }) => { + const result = Bun.spawnSync([command, ...args], { + cwd: execOptions?.cwd ?? process.cwd(), + stdout: "pipe", + stderr: "pipe", + }); + return { + code: result.exitCode, + stdout: Buffer.from(result.stdout).toString("utf8"), + stderr: Buffer.from(result.stderr).toString("utf8"), + }; + }, + getActiveTools: () => [...activeTools], + setActiveTools: async (toolNames: string[]) => { + setActiveToolsCalls.push([...toolNames]); + activeTools.splice(0, activeTools.length, ...toolNames); + }, + } as unknown as ExtensionAPI; + return { activeTools, api, appendEntries, setActiveToolsCalls }; +} + +function expectRunDetails(details: unknown): RunDetails { + if (!details || typeof details !== "object" || !("benchmarkLogPath" in details)) { + throw new Error("Expected run details"); + } + return details as RunDetails; +} + +describe("autoresearch tools", () => { + const tempDirs: string[] = []; + + afterEach(() => { + for (const dir of tempDirs.splice(0)) { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + it("writes durable benchmark/check artifacts and run metadata", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + checksScript: ["#!/usr/bin/env bash", "set -euo pipefail", "echo checks ok"].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "call-1", + { command: "bash autoresearch.sh", timeout_seconds: 5, checks_timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + const details = expectRunDetails(result.details); + + expect(details.runNumber).toBe(1); + expect(details.parsedPrimary).toBe(10); + expect(details.parsedMetrics).toEqual({ memory_mb: 32, runtime_ms: 10 }); + expect(details.benchmarkLogPath).toBe(path.join(details.runDirectory, "benchmark.log")); + expect(fs.existsSync(details.benchmarkLogPath)).toBe(true); + expect(fs.existsSync(details.checksLogPath ?? "")).toBe(true); + + const runJson = JSON.parse(fs.readFileSync(path.join(details.runDirectory, "run.json"), "utf8")) as { + completedAt?: string; + parsedPrimary?: number; + checks?: { passed?: boolean }; + }; + expect(runJson.completedAt).toEqual(expect.any(String)); + expect(runJson.parsedPrimary).toBe(10); + expect(runJson.checks?.passed).toBe(true); + }); + + it("ignores incomplete run artifacts until the benchmark has actually finished", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + runNumber: 1, + startedAt: new Date().toISOString(), + }), + ); + + const pendingRun = await readPendingRunSummary(dir); + expect(pendingRun).toBeNull(); + }); + + it("persists init_experiment config metadata from autoresearch.md", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "- secondary metrics: memory_mb, tokens", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "init-1", + { + name: "Reduce runtime variance", + metric_name: "runtime_ms", + metric_unit: "ms", + direction: "lower", + benchmark_command: "bash autoresearch.sh", + scope_paths: ["src"], + off_limits: ["src/generated"], + constraints: ["keep behavior stable"], + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Experiment initialized: Reduce runtime variance"), + }); + expect(runtime.state.secondaryMetrics).toEqual([ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]); + + const configEntry = JSON.parse(fs.readFileSync(path.join(dir, "autoresearch.jsonl"), "utf8").trim()) as { + benchmarkCommand?: string; + constraints?: string[]; + offLimits?: string[]; + scopePaths?: string[]; + secondaryMetrics?: string[]; + segmentFingerprint?: string; + }; + expect(configEntry.benchmarkCommand).toBe("bash autoresearch.sh"); + expect(configEntry.secondaryMetrics).toEqual(["memory_mb", "tokens"]); + expect(configEntry.scopePaths).toEqual(["src"]); + expect(configEntry.offLimits).toEqual(["src/generated"]); + expect(configEntry.constraints).toEqual(["keep behavior stable"]); + expect(configEntry.segmentFingerprint).toBe(createFingerprint(dir)); + }); + + it("rejects init_experiment when the passed contract no longer matches autoresearch.md", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "init-2", + { + name: "Mismatch", + metric_name: "runtime_ms", + metric_unit: "ms", + direction: "lower", + benchmark_command: "bash autoresearch.sh", + scope_paths: ["src"], + off_limits: ["src/other-generated"], + constraints: ["keep behavior stable"], + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("off_limits do not match autoresearch.md"), + }); + expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false); + }); + + it("refuses to start a new benchmark while a previous run artifact is still unlogged", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ command: "bash autoresearch.sh", exitCode: 0, parsedPrimary: 10, runNumber: 1 }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-1b", + { command: "bash autoresearch.sh", timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("has not been logged yet"), + }); + }); + + it("refuses to run when the current segment fingerprint is stale", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + fs.writeFileSync( + path.join(dir, "autoresearch.sh"), + ["#!/usr/bin/env bash", "set -euo pipefail", "echo METRIC runtime_ms=9"].join("\n"), + ); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-2", + { command: "bash autoresearch.sh", timeout_seconds: 5 }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Re-run init_experiment"), + }); + }); + + it("times out checks asynchronously and preserves the checks log", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + checksScript: ["#!/usr/bin/env bash", "set -euo pipefail", "sleep 2", "echo done"].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-3", + { command: "bash autoresearch.sh", timeout_seconds: 5, checks_timeout_seconds: 0.1 }, + undefined, + undefined, + createContext(dir), + ); + const details = expectRunDetails(result.details); + + expect(details.checksTimedOut).toBe(true); + expect(fs.existsSync(details.checksLogPath ?? "")).toBe(true); + }); + + it("honors user aborts while the experiment is running", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + benchmarkScript: ["#!/usr/bin/env bash", "set -euo pipefail", "sleep 5", "echo METRIC runtime_ms=10"].join( + "\n", + ), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + + const tool = createRunExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const controller = new AbortController(); + setTimeout(() => controller.abort(), 100); + + await expect( + tool.execute( + "call-4", + { command: "bash autoresearch.sh", timeout_seconds: 10 }, + controller.signal, + undefined, + createContext(dir), + ), + ).rejects.toThrow("aborted"); + expect(fs.existsSync(path.join(dir, ".autoresearch", "runs", "0001", "run.json"))).toBe(true); + }); + + it("commits only in-scope changes and excludes autoresearch local state", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src/in-scope.ts", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "- keep behavior stable", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 1;\n"); + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 2;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 3;\n"); + fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n\n- focus on in-scope edits\n"); + fs.writeFileSync(path.join(dir, "autoresearch.jsonl"), '{"type":"run"}\n'); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src/in-scope.ts"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.constraints = ["keep behavior stable"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + const runDirectory = path.join(dir, ".autoresearch", "runs", "0001"); + runtime.lastRunArtifactDir = runDirectory; + runtime.lastRunNumber = 1; + runtime.lastRunDuration = 1.2; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1.2, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory, + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-5", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Improve in scope", + asi: { hypothesis: "inline the hot path" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Logged run #1: keep"), + }); + const committedPaths = await $`git show --name-only --pretty=format: HEAD`.cwd(dir).text(); + expect(committedPaths).toContain("src/in-scope.ts"); + expect(committedPaths).toContain("autoresearch.program.md"); + expect(committedPaths).not.toContain("autoresearch.jsonl"); + expect(committedPaths).not.toContain(".autoresearch"); + + const runJson = JSON.parse(fs.readFileSync(path.join(runDirectory, "run.json"), "utf8")) as { + status?: string; + }; + expect(runJson.status).toBe("keep"); + }); + + it("rejects keep when an out-of-scope file is dirty", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src/in-scope.ts", + "", + "## Off Limits", + "", + "## Constraints", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 1;\n"); + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 2;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 99;\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src/in-scope.ts"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-6", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail", + asi: { hypothesis: "touch wrong file" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("outside Files in Scope"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("rejects keep when a dirty path is listed under Off Limits", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "", + ].join("\n"), + }); + fs.mkdirSync(path.join(dir, "src", "generated"), { recursive: true }); + fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 1;\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 2;\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-7", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail", + asi: { hypothesis: "touch forbidden path" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Off Limits"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("rejects keep when the metric is worse than the current best kept run", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0003", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 3, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["autoresearch.md"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.state.results = [ + { + runNumber: 1, + commit: "aaaaaaa", + metric: 10, + metrics: {}, + status: "keep", + description: "baseline", + timestamp: 1, + segment: 0, + confidence: null, + }, + { + runNumber: 2, + commit: "bbbbbbb", + metric: 8, + metrics: {}, + status: "keep", + description: "winner", + timestamp: 2, + segment: 0, + confidence: null, + }, + ]; + runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "0003"); + runtime.lastRunNumber = 3; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0003"), + runNumber: 3, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-best", + { + commit: "initial", + metric: 9, + status: "keep", + description: "regression from best", + asi: { hypothesis: "try a weaker variant" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Current best: 8"), + }); + expect(runtime.state.results).toHaveLength(2); + }); + + it("requires failed benchmarks to be logged as crash", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 1, + runNumber: 1, + timedOut: false, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: null, + parsedPrimary: null, + passed: false, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: { + exec: async () => ({ code: 0, stderr: "", stdout: "autoresearch/test-20260323\n" }), + } as unknown as ExtensionAPI, + }); + const result = await tool.execute( + "call-status-crash", + { + commit: "initial", + metric: 0, + status: "discard", + description: "wrong status", + asi: { + hypothesis: "broken attempt", + rollback_reason: "benchmark failed", + next_action_hint: "fix the crash first", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Log it as crash"), + }); + }); + + it("requires failed checks to be logged as checks_failed", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + checks: { durationSeconds: 1, passed: false, timedOut: false }, + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 1, + checksPass: false, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: null, + parsedPrimary: 10, + passed: false, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: { + exec: async () => ({ code: 0, stderr: "", stdout: "autoresearch/test-20260323\n" }), + } as unknown as ExtensionAPI, + }); + const result = await tool.execute( + "call-status-checks", + { + commit: "initial", + metric: 10, + status: "crash", + description: "wrong checks status", + asi: { + hypothesis: "checks regressed", + rollback_reason: "test suite failed", + next_action_hint: "inspect failing checks", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Log it as checks_failed"), + }); + }); + + it("persists autoresearch shutdown when the max iteration cap is reached", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.autoresearchMode = true; + runtime.goal = "reduce runtime"; + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["autoresearch.md"]; + runtime.state.maxExperiments = 1; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "0001"); + runtime.lastRunNumber = 1; + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const managedApi = createManagedGitApi({ + activeTools: ["read", "init_experiment", "run_experiment", "log_experiment"], + }); + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: managedApi.api, + }); + const result = await tool.execute( + "call-max", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Baseline", + asi: { hypothesis: "record baseline" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Autoresearch mode is now off"), + }); + expect(runtime.autoresearchMode).toBe(false); + expect(managedApi.appendEntries).toContainEqual({ + customType: "autoresearch-control", + data: { mode: "off", goal: "reduce runtime" }, + }); + expect(managedApi.setActiveToolsCalls).toEqual([["read"]]); + }); + + it("rejects keep when a rename touches an off-limits source path", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir, { + contract: [ + "# Autoresearch", + "", + "## Benchmark", + "- command: bash autoresearch.sh", + "- primary metric: runtime_ms", + "- metric unit: ms", + "- direction: lower", + "", + "## Files in Scope", + "- src", + "", + "## Off Limits", + "- src/generated", + "", + "## Constraints", + "", + ].join("\n"), + }); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.offLimits = ["src/generated"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const api = { + exec: async (command: string, args: string[]) => { + if (command !== "git") return { code: 1, stderr: "unexpected", stdout: "" }; + if (args[0] === "status") { + return { code: 0, stderr: "", stdout: "R src/generated/index.ts\0src/index.ts\0" }; + } + return { code: 1, stderr: `unexpected git args: ${args.join(" ")}`, stdout: "" }; + }, + } as unknown as ExtensionAPI; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: api, + }); + const result = await tool.execute( + "call-8", + { + commit: "initial", + metric: 9, + status: "keep", + description: "Should fail on rename", + asi: { hypothesis: "rename generated file" }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Off Limits"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("removes ignored experiment artifacts on discard while preserving autoresearch control files", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + fs.writeFileSync(path.join(dir, ".gitignore"), "tmp-artifact/\n"); + fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n"); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + + fs.mkdirSync(path.join(dir, "tmp-artifact"), { recursive: true }); + fs.writeFileSync(path.join(dir, "tmp-artifact", "result.txt"), "temporary benchmark output\n"); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + exitCode: 0, + parsedMetrics: { runtime_ms: 10 }, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.scopePaths = ["src"]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: null, + parsedAsi: null, + parsedMetrics: { runtime_ms: 10 }, + parsedPrimary: 10, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-discard", + { + commit: "initial", + metric: 10, + status: "discard", + description: "Discard noisy run", + asi: { + hypothesis: "investigate cache behavior", + rollback_reason: "ignored artifact should be cleaned", + next_action_hint: "try a cleaner setup", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Logged run #1: discard"), + }); + expect(fs.existsSync(path.join(dir, "tmp-artifact"))).toBe(false); + expect(fs.existsSync(path.join(dir, "autoresearch.md"))).toBe(true); + expect(fs.existsSync(path.join(dir, "autoresearch.program.md"))).toBe(true); + }); +}); From 2c93655796de2f4f59e379e9ace21a41c3de9ec8 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 23 Mar 2026 02:16:39 +0100 Subject: [PATCH 19/22] feat(autoresearch): added auto-resume, path validation, and security guards - Added auto-resume mechanism with state tracking to automatically resume pending experiment runs and prevent duplicate resumptions. - Added contract path validation to reject unsafe path specifications with absolute paths and parent directory traversal attempts. - Added secondary metrics input to autoresearch setup flow for specifying tradeoff metrics alongside primary objectives. - Enhanced command parsing with shell operator detection to reject piped, redirected, or chained autoresearch.sh commands. - Added prototype pollution guards in object cloning functions to prevent injection via __proto__, constructor, and prototype keys. - Fixed boundary duplication warnings in hashline detection to properly report multiple overlapping hashline references. --- packages/coding-agent/CHANGELOG.md | 25 +- .../src/autoresearch/command-initialize.md | 8 +- .../src/autoresearch/command-resume.md | 2 +- .../coding-agent/src/autoresearch/contract.ts | 16 +- .../coding-agent/src/autoresearch/helpers.ts | 34 +- .../coding-agent/src/autoresearch/index.ts | 29 +- .../coding-agent/src/autoresearch/prompt.md | 15 +- .../src/autoresearch/resume-message.md | 2 +- .../coding-agent/src/autoresearch/state.ts | 10 +- .../src/autoresearch/tools/init-experiment.ts | 32 ++ .../src/autoresearch/tools/log-experiment.ts | 15 + .../src/autoresearch/tools/run-experiment.ts | 4 + .../coding-agent/src/autoresearch/types.ts | 2 + packages/coding-agent/src/patch/hashline.ts | 2 +- .../src/prompts/tools/hashline.md | 2 +- .../test/agent-session-concurrent.test.ts | 33 +- .../test/autoresearch-state.test.ts | 145 ++++++- .../test/autoresearch-tools.test.ts | 354 ++++++++++++++++++ .../coding-agent/test/core/hashline.test.ts | 8 +- 19 files changed, 698 insertions(+), 40 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8d165f70b..b55c0720e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Changed hashline edit schema from flat `op`/`pos`/`end`/`lines` fields to structured `loc`/`content` format with location-specific objects @@ -14,6 +13,14 @@ ### Added +- Added prompt for tradeoff metrics during autoresearch setup to collect secondary metrics alongside primary metric +- Added validation of contract path specifications to reject absolute paths and parent directory references +- Added stricter benchmark command validation in `isAutoresearchShCommand()` to reject chained commands, pipes, and redirects +- Added protection against prototype pollution in ASI data and metric cloning by filtering `__proto__`, `constructor`, and `prototype` keys +- Added `autoResumeArmed` flag to track when autoresearch should automatically resume pending runs +- Added `lastAutoResumePendingRunNumber` to prevent duplicate auto-resume prompts for the same pending run +- Added `git clean -X` invocation during failed experiment rollback to remove ignored build artifacts +- Added validation to reject `init_experiment` when a previous run is still pending and unlogged - Added autoresearch contract system for validating benchmark commands, metrics, scope paths, off-limits paths, and constraints with fingerprint tracking to detect configuration drift - Added `autoresearch.program.md` support for repo-local playbook overlays that guide session strategy while preserving `autoresearch.md` as source of truth - Added pending run artifact tracking and recovery to resume incomplete experiments from `.autoresearch/runs/` directory with run numbers and benchmark logs @@ -62,6 +69,19 @@ ### Changed +- Changed `isAutoresearchShCommand()` to use proper command-line argument parsing instead of regex, improving accuracy for complex shell invocations +- Changed autoresearch initialization prompt to display collected tradeoff metrics in the setup summary +- Changed `command-initialize.md` template to include guidance on preflight requirements, comparability invariants, and marking measurement-critical files as off-limits +- Changed `command-initialize.md` to instruct users to write or update `autoresearch.program.md` with durable heuristics and repo-specific strategy +- Changed autoresearch resume guidance to emphasize continuing on the current protected branch rather than switching branches +- Changed autoresearch prompt to clarify that `autoresearch.md` holds durable conclusions while `autoresearch.ideas.md` is the scratch backlog +- Changed autoresearch prompt guidance to require stable measurement harness and fixed benchmark inputs unless intentionally starting a new segment +- Changed autoresearch prompt to recommend keeping equal or near-equal results when they materially simplify implementation +- Changed `init_experiment` to reset pending run state (checks, duration, ASI, artifact directory) when initializing a new segment +- Changed `log_experiment` to set `autoResumeArmed` flag after successfully logging a run to enable auto-resume on next agent turn +- Changed `run_experiment` to set `autoResumeArmed` flag and update dashboard after completing a run +- Changed auto-resume logic to only prompt when a new pending run exists or when `autoResumeArmed` is explicitly set, preventing duplicate prompts +- Changed path normalization in contract validation to use `path.posix.normalize()` for consistent path handling - Changed autoresearch initialization to collect and validate benchmark command, metric definition, scope paths, off-limits list, and constraints before `init_experiment` - Changed `init_experiment` to require exact benchmark command, metric definition, scope, off-limits, and constraints matching collected contract - Changed `log_experiment` to record run number, benchmark command, scope paths, off-limits list, constraints, and segment fingerprint with each result @@ -111,6 +131,9 @@ ### Fixed +- Fixed boundary duplication warnings to always display when replacement lines match the next surviving line, even when auto-correction is disabled +- Fixed secondary metrics validation to properly reject missing configured metrics and new metrics without force flag +- Fixed ASI data cloning to prevent prototype pollution attacks by filtering reserved property names - Fixed autoresearch resume to detect and recover pending run artifacts that were left unlogged from previous sessions - Fixed dashboard overlay to display when running experiment even with zero completed results - Fixed tab character rendering in dashboard command display and tool output summaries diff --git a/packages/coding-agent/src/autoresearch/command-initialize.md b/packages/coding-agent/src/autoresearch/command-initialize.md index 9986a844b..1e7939d42 100644 --- a/packages/coding-agent/src/autoresearch/command-initialize.md +++ b/packages/coding-agent/src/autoresearch/command-initialize.md @@ -10,6 +10,8 @@ Collected setup: - primary metric: `{{metric_name}}` - metric unit: `{{metric_unit}}` - direction: `{{direction}}` +- tradeoff metrics: +{{{secondary_metrics_block}}} - files in scope: {{{scope_paths_block}}} - off limits: @@ -21,8 +23,10 @@ Explain briefly what autoresearch will do in this repository, then initialize th Your first actions: - write `autoresearch.md` -- record the collected benchmark command, primary metric, metric unit, direction, scope, off-limits list, and constraints in `autoresearch.md` -- optionally write `autoresearch.program.md` when a repo-local playbook would help future resume quality +- record the collected benchmark command, primary metric, metric unit, direction, tradeoff metrics, scope, off-limits list, and constraints in `autoresearch.md` +- add a short preflight section in `autoresearch.md` covering prerequisites, one-time setup, and the comparability invariant that must stay fixed across runs +- explicitly mark the ground-truth evaluator, fixed datasets, and other measurement-critical files as off-limits or hard constraints when they define the benchmark contract +- write or update `autoresearch.program.md` when you learn durable heuristics, failure patterns, or repo-specific strategy that future resume turns should inherit - define the benchmark entrypoint in `autoresearch.sh` - optionally add `autoresearch.checks.sh` if correctness or quality needs a hard gate - run `init_experiment` with the exact collected benchmark command, metric definition, scope paths, off-limits list, and constraints diff --git a/packages/coding-agent/src/autoresearch/command-resume.md b/packages/coding-agent/src/autoresearch/command-resume.md index 3dd0030a4..e71543cb1 100644 --- a/packages/coding-agent/src/autoresearch/command-resume.md +++ b/packages/coding-agent/src/autoresearch/command-resume.md @@ -13,5 +13,5 @@ Additional context from the user: Use the notes as the source of truth for the current direction, scope, and constraints. - inspect recent git history for context - inspect `autoresearch.jsonl` if it exists -- continue the most promising unfinished branch +- continue the most promising unfinished direction on the current protected branch - keep iterating until interrupted or until the configured iteration cap is reached diff --git a/packages/coding-agent/src/autoresearch/contract.ts b/packages/coding-agent/src/autoresearch/contract.ts index 45d4c4e46..c5b8b87c9 100644 --- a/packages/coding-agent/src/autoresearch/contract.ts +++ b/packages/coding-agent/src/autoresearch/contract.ts @@ -63,6 +63,16 @@ export function validateAutoresearchContract(contract: AutoresearchContract): st if (contract.scopePaths.length === 0) { errors.push("Files in Scope must contain at least one path in autoresearch.md."); } + for (const scopePath of contract.scopePaths) { + if (isUnsafeContractPathSpec(scopePath)) { + errors.push(`Files in Scope contains an invalid path: ${scopePath}`); + } + } + for (const offLimitsPath of contract.offLimits) { + if (isUnsafeContractPathSpec(offLimitsPath)) { + errors.push(`Off Limits contains an invalid path: ${offLimitsPath}`); + } + } return errors; } @@ -151,7 +161,7 @@ export function normalizeAutoresearchList(values: readonly string[]): string[] { } export function normalizeContractPathSpec(value: string): string { - const normalized = value.trim().replaceAll("\\", "/"); + const normalized = path.posix.normalize(value.trim().replaceAll("\\", "/")); if (normalized === "." || normalized === "./") return "."; return normalized.replace(/^\.\/+/, "").replace(/\/+$/, ""); } @@ -316,3 +326,7 @@ function parseSecondaryMetrics(value: string | undefined): string[] { .filter(Boolean), ); } + +function isUnsafeContractPathSpec(value: string): boolean { + return path.posix.isAbsolute(value) || value === ".." || value.startsWith("../"); +} diff --git a/packages/coding-agent/src/autoresearch/helpers.ts b/packages/coding-agent/src/autoresearch/helpers.ts index 20f2ae0f7..e278d3631 100644 --- a/packages/coding-agent/src/autoresearch/helpers.ts +++ b/packages/coding-agent/src/autoresearch/helpers.ts @@ -1,6 +1,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; +import { parseCommandArgs } from "../utils/command-args"; import type { ASIData, ASIValue, @@ -185,8 +186,36 @@ export function isAutoresearchShCommand(command: string): boolean { previous = normalized; normalized = normalized.replace(/^(?:env|time|nice|nohup)(?:\s+-\S+(?:\s+\d+)?)?\s+/, ""); } + if (/[;&|<>]/.test(normalized)) { + return false; + } - return /^(?:(?:bash|sh)\s+(?:-\w+\s+)*)?(?:\.\/|\/[\w/.-]*\/)?autoresearch\.sh(?:\s|$)/.test(normalized); + const tokens = parseCommandArgs(normalized); + if (tokens.length === 0) return false; + + let index = 0; + if (tokens[index] === "bash" || tokens[index] === "sh") { + index += 1; + while (index < tokens.length && tokens[index]?.startsWith("-")) { + if (tokens[index]?.includes("c")) { + return false; + } + index += 1; + } + } + + const scriptToken = tokens[index]; + if (!scriptToken || !/^(?:\.\/|\/[\w/.-]*\/)?autoresearch\.sh$/.test(scriptToken)) { + return false; + } + + for (const token of tokens.slice(index + 1)) { + if (token === "&&" || token === "||" || token === ";" || token === "|" || token === ">" || token === "<") { + return false; + } + } + + return true; } export function isBetter(current: number, best: number, direction: MetricDirection): boolean { @@ -380,6 +409,7 @@ function cloneNumericMetricMap(value: unknown): NumericMetricMap | null { const metrics = value as { [key: string]: unknown }; const clone: NumericMetricMap = {}; for (const [key, entryValue] of Object.entries(metrics)) { + if (DENIED_KEY_NAMES.has(key)) continue; if (typeof entryValue === "number" && Number.isFinite(entryValue)) { clone[key] = entryValue; } @@ -392,6 +422,7 @@ function cloneAsiData(value: unknown): ASIData | null { const candidate = value as { [key: string]: unknown }; const clone: ASIData = {}; for (const [key, entryValue] of Object.entries(candidate)) { + if (DENIED_KEY_NAMES.has(key)) continue; const sanitized = clonePendingAsiValue(entryValue); if (sanitized !== undefined) { clone[key] = sanitized; @@ -415,6 +446,7 @@ function clonePendingAsiValue(value: unknown): ASIValue | undefined { const candidate = value as { [key: string]: unknown }; const clone: { [key: string]: ASIValue } = {}; for (const [key, entryValue] of Object.entries(candidate)) { + if (DENIED_KEY_NAMES.has(key)) continue; const sanitized = clonePendingAsiValue(entryValue); if (sanitized !== undefined) { clone[key] = sanitized; diff --git a/packages/coding-agent/src/autoresearch/index.ts b/packages/coding-agent/src/autoresearch/index.ts index 4ddcbd2c2..8173b12d3 100644 --- a/packages/coding-agent/src/autoresearch/index.ts +++ b/packages/coding-agent/src/autoresearch/index.ts @@ -43,6 +43,7 @@ interface AutoresearchSetupInput { metricName: string; metricUnit: string; direction: "lower" | "higher"; + secondaryMetrics: string[]; scopePaths: string[]; offLimits: string[]; constraints: string[]; @@ -65,6 +66,8 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.state.maxExperiments = readMaxExperiments(ctx.cwd); runtime.goal = control.goal; runtime.autoresearchMode = control.autoresearchMode; + runtime.autoResumeArmed = false; + runtime.lastAutoResumePendingRunNumber = null; runtime.lastRunSummary = await readPendingRunSummary(workDir, loggedRunNumbers); runtime.lastRunChecks = summaryToChecks(runtime.lastRunSummary); runtime.lastRunDuration = runtime.lastRunSummary?.durationSeconds ?? null; @@ -94,7 +97,9 @@ export const createAutoresearchExtension: ExtensionFactory = api => { ): void => { const runtime = getRuntime(ctx); runtime.autoresearchMode = enabled; + runtime.autoResumeArmed = false; runtime.goal = goal; + runtime.lastAutoResumePendingRunNumber = null; api.appendEntry("autoresearch-control", goal ? { mode, goal } : { mode }); }; @@ -251,6 +256,7 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.state.metricName = setup.metricName; runtime.state.metricUnit = setup.metricUnit; runtime.state.bestDirection = setup.direction; + runtime.state.secondaryMetrics = setup.secondaryMetrics.map(name => ({ name, unit: "" })); runtime.state.benchmarkCommand = setup.benchmarkCommand; runtime.state.scopePaths = [...setup.scopePaths]; runtime.state.offLimits = [...setup.offLimits]; @@ -267,6 +273,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => { metric_name: setup.metricName, metric_unit: setup.metricUnit, direction: setup.direction, + has_secondary_metrics: setup.secondaryMetrics.length > 0, + secondary_metrics: setup.secondaryMetrics, + secondary_metrics_block: formatBulletBlock( + setup.secondaryMetrics, + value => ` - \`${value}\``, + " - `(none)`", + ), scope_paths: setup.scopePaths, scope_paths_block: formatBulletBlock(setup.scopePaths, value => ` - \`${value}\``), has_off_limits: setup.offLimits.length > 0, @@ -315,7 +328,10 @@ export const createAutoresearchExtension: ExtensionFactory = api => { dashboard.updateWidget(ctx, runtime); dashboard.requestRender(); if (!runtime.autoresearchMode) return; - if (ctx.hasPendingMessages()) return; + if (ctx.hasPendingMessages()) { + runtime.autoResumeArmed = false; + return; + } const workDir = resolveWorkDir(ctx.cwd); const pendingRun = runtime.lastRunSummary ?? @@ -324,6 +340,13 @@ export const createAutoresearchExtension: ExtensionFactory = api => { runtime.lastRunChecks = summaryToChecks(pendingRun); runtime.lastRunDuration = pendingRun?.durationSeconds ?? runtime.lastRunDuration; runtime.lastRunAsi = pendingRun?.parsedAsi ?? runtime.lastRunAsi; + const shouldResumePendingRun = + pendingRun !== null && runtime.lastAutoResumePendingRunNumber !== pendingRun.runNumber; + if (!shouldResumePendingRun && !runtime.autoResumeArmed) { + return; + } + runtime.autoResumeArmed = false; + runtime.lastAutoResumePendingRunNumber = pendingRun?.runNumber ?? null; const autoresearchMdPath = path.join(workDir, "autoresearch.md"); const ideasPath = path.join(workDir, "autoresearch.ideas.md"); api.sendMessage( @@ -459,6 +482,9 @@ async function promptForAutoresearchSetup( return undefined; } + const secondaryMetricsInput = await ctx.ui.input("Tradeoff Metrics", ""); + if (secondaryMetricsInput === undefined) return undefined; + const scopePathsInput = await ctx.ui.input("Files in Scope", "packages/coding-agent/src/autoresearch"); if (scopePathsInput === undefined) return undefined; const scopePaths = splitSetupList(scopePathsInput); @@ -478,6 +504,7 @@ async function promptForAutoresearchSetup( metricName, metricUnit, direction: normalizedDirection, + secondaryMetrics: splitSetupList(secondaryMetricsInput), scopePaths, offLimits: splitSetupList(offLimitsInput), constraints: splitSetupList(constraintsInput), diff --git a/packages/coding-agent/src/autoresearch/prompt.md b/packages/coding-agent/src/autoresearch/prompt.md index c02c20f13..185edcbfe 100644 --- a/packages/coding-agent/src/autoresearch/prompt.md +++ b/packages/coding-agent/src/autoresearch/prompt.md @@ -71,13 +71,17 @@ An unlogged run artifact exists at `{{pending_run_directory}}`. - Read the relevant source files. - Identify the true bottleneck or quality constraint. - Check existing scripts, benchmark harnesses, and config files. + - Verify prerequisites, one-time setup, and benchmark inputs before the first run of a segment. 2. Keep your notes in `autoresearch.md`. - - Record the goal, the benchmark command, the primary metric, important secondary metrics, the files in scope, hard constraints, and the running ideas backlog. + - Record the goal, the benchmark command, the primary metric, important secondary metrics, the files in scope, hard constraints, preflight requirements, and the benchmark comparability invariant. - Update the notes whenever the strategy changes. + - Keep durable conclusions in `autoresearch.md`. + - Use `autoresearch.ideas.md` for deferred experiment ideas that are promising but not active yet. 3. Use `autoresearch.sh` as the canonical benchmark entrypoint. - If it does not exist yet, create it. - Make it print structured metric lines in the form `METRIC name=value`. - Use the same workload every run unless you intentionally re-initialize with a new segment. + - Keep the measurement harness, evaluator, and fixed benchmark inputs stable unless you intentionally start a new segment and document the change. 4. Initialize the loop with `init_experiment` before the first logged run of a segment. 5. Run a baseline first. - Establish the baseline metric before attempting optimizations. @@ -98,7 +102,8 @@ An unlogged run artifact exists at `{{pending_run_directory}}`. - Use ASI to capture what you learned, not just what you changed. 9. Prefer simpler wins. - Remove dead ends. - - Do not keep complexity that does not move the metric. + - Keep equal or near-equal results when they materially simplify the implementation. + - Do not keep ugly complexity for tiny gains unless the payoff is clearly worth it. - Do not thrash between unrelated ideas without writing down the conclusion. 10. When confidence is low, confirm. - The dashboard confidence score compares the best observed improvement against the observed noise floor. @@ -116,6 +121,8 @@ Your benchmark script SHOULD: - print secondary metrics as additional `METRIC name=value` lines - avoid extra randomness when possible - use repeated samples and median-style summaries for fast benchmarks +- preserve the comparability invariant for the current segment +- keep the ground-truth evaluator and fixed benchmark inputs unchanged unless the segment is explicitly re-initialized ### Notes file template @@ -182,7 +189,7 @@ Resume from the existing notes: - read `autoresearch.md` - inspect recent git history - inspect `autoresearch.jsonl` -- continue from the most promising unfinished branch +- continue from the most promising unfinished direction on the current protected branch {{else}} ### Initial setup @@ -215,6 +222,6 @@ Treat failing checks as a failed experiment: `autoresearch.ideas.md` exists at `{{ideas_path}}`. -Use it to keep promising but deferred experiments. Prune stale ideas when they are disproven or superseded. +Use it to keep promising but deferred experiments. `autoresearch.md` should hold durable conclusions; `autoresearch.ideas.md` is the scratch backlog. Prune stale ideas when they are disproven or superseded. {{/if}} diff --git a/packages/coding-agent/src/autoresearch/resume-message.md b/packages/coding-agent/src/autoresearch/resume-message.md index 64c4816f3..31052bb78 100644 --- a/packages/coding-agent/src/autoresearch/resume-message.md +++ b/packages/coding-agent/src/autoresearch/resume-message.md @@ -10,7 +10,7 @@ Continue the autoresearch loop now. {{/if}} - Continue from the most promising unfinished direction. {{#if has_ideas}} -- Review `autoresearch.ideas.md` for promising next steps and prune stale items. +- Review `autoresearch.ideas.md` for deferred next steps and prune stale items. {{/if}} - Keep iterating until interrupted or until the configured iteration cap is reached. - Preserve correctness and do not game the benchmark. diff --git a/packages/coding-agent/src/autoresearch/state.ts b/packages/coding-agent/src/autoresearch/state.ts index 725a60715..9ab05a60d 100644 --- a/packages/coding-agent/src/autoresearch/state.ts +++ b/packages/coding-agent/src/autoresearch/state.ts @@ -41,7 +41,9 @@ export function createExperimentState(): ExperimentState { export function createSessionRuntime(): AutoresearchRuntime { return { autoresearchMode: false, + autoResumeArmed: false, dashboardExpanded: false, + lastAutoResumePendingRunNumber: null, lastRunChecks: null, lastRunDuration: null, lastRunAsi: null, @@ -341,6 +343,7 @@ function cloneNumericMetrics(value: unknown): NumericMetricMap { const metrics = value as { [key: string]: unknown }; const clone: NumericMetricMap = {}; for (const [key, entryValue] of Object.entries(metrics)) { + if (key === "__proto__" || key === "constructor" || key === "prototype") continue; if (typeof entryValue === "number" && Number.isFinite(entryValue)) { clone[key] = entryValue; } @@ -363,7 +366,12 @@ function hydrateMetricDefs(metricNames: string[] | undefined): MetricDef[] { function cloneAsi(value: unknown): ExperimentResult["asi"] { if (typeof value !== "object" || value === null) return undefined; - return structuredClone(value) as ExperimentResult["asi"]; + const clone: { [key: string]: unknown } = {}; + for (const [key, entryValue] of Object.entries(value)) { + if (key === "__proto__" || key === "constructor" || key === "prototype") continue; + clone[key] = structuredClone(entryValue); + } + return clone as ExperimentResult["asi"]; } function parseControlEntry(value: unknown): AutoresearchControlEntryData | null { diff --git a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts index 5e81714ba..cf8a0f63f 100644 --- a/packages/coding-agent/src/autoresearch/tools/init-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/init-experiment.ts @@ -17,6 +17,7 @@ import { inferMetricUnitFromName, isAutoresearchShCommand, readMaxExperiments, + readPendingRunSummary, resolveWorkDir, validateWorkDir, } from "../helpers"; @@ -85,6 +86,19 @@ export function createInitExperimentTool( const state = runtime.state; const isReinitializing = state.results.length > 0; const workDir = resolveWorkDir(ctx.cwd); + const pendingRun = await readPendingRunSummary(workDir, collectLoggedRunNumbers(state.results)); + if (pendingRun) { + return { + content: [ + { + type: "text", + text: + `Error: run #${pendingRun.runNumber} has not been logged yet. ` + + "Call log_experiment before re-initializing the current segment.", + }, + ], + }; + } const contractResult = readAutoresearchContract(workDir); const scriptSnapshot = loadAutoresearchScriptSnapshot(workDir); const errors = [...contractResult.errors, ...scriptSnapshot.errors]; @@ -241,6 +255,14 @@ export function createInitExperimentTool( } runtime.autoresearchMode = true; + runtime.autoResumeArmed = true; + runtime.lastAutoResumePendingRunNumber = null; + runtime.lastRunChecks = null; + runtime.lastRunDuration = null; + runtime.lastRunAsi = null; + runtime.lastRunArtifactDir = null; + runtime.lastRunNumber = null; + runtime.lastRunSummary = null; options.dashboard.updateWidget(ctx, runtime); options.dashboard.requestRender(); @@ -276,3 +298,13 @@ export function createInitExperimentTool( function renderInitCall(name: string, theme: Theme): string { return `${theme.fg("toolTitle", theme.bold("init_experiment"))} ${theme.fg("accent", truncateToWidth(replaceTabs(name), 100))}`; } + +function collectLoggedRunNumbers(results: ExperimentState["results"]): Set { + const runNumbers = new Set(); + for (const result of results) { + if (result.runNumber !== null) { + runNumbers.add(result.runNumber); + } + } + return runNumbers; +} diff --git a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts index 1a25db2ef..ec9f6caee 100644 --- a/packages/coding-agent/src/autoresearch/tools/log-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/log-experiment.ts @@ -304,6 +304,8 @@ export function createLogExperimentTool( runtime.lastRunArtifactDir = null; runtime.lastRunNumber = null; runtime.lastRunSummary = null; + runtime.autoResumeArmed = true; + runtime.lastAutoResumePendingRunNumber = null; const currentSegmentRuns = currentResults(state.results, state.currentSegment).length; const text = buildLogText(state, experiment, currentSegmentRuns, wallClockSeconds, gitNote); @@ -364,10 +366,12 @@ function buildSecondaryMetrics( ): NumericMetricMap { const merged: NumericMetricMap = {}; for (const [name, value] of Object.entries(parsedMetrics ?? {})) { + if (name === "__proto__" || name === "constructor" || name === "prototype") continue; if (name === primaryMetricName) continue; merged[name] = value; } for (const [name, value] of Object.entries(cloneMetrics(overrides))) { + if (name === "__proto__" || name === "constructor" || name === "prototype") continue; merged[name] = value; } return merged; @@ -377,6 +381,7 @@ function sanitizeAsi(value: { [key: string]: unknown } | undefined): ASIData | u if (!value) return undefined; const result: ASIData = {}; for (const [key, entryValue] of Object.entries(value)) { + if (key === "__proto__" || key === "constructor" || key === "prototype") continue; const sanitized = sanitizeAsiValue(entryValue); if (sanitized !== undefined) { result[key] = sanitized; @@ -398,6 +403,7 @@ function sanitizeAsiValue(value: unknown): ASIData[string] | undefined { const objectValue = value as { [key: string]: unknown }; const result: ASIData = {}; for (const [key, entryValue] of Object.entries(objectValue)) { + if (key === "__proto__" || key === "constructor" || key === "prototype") continue; const sanitized = sanitizeAsiValue(entryValue); if (sanitized !== undefined) { result[key] = sanitized; @@ -567,6 +573,10 @@ async function revertFailedExperiment( { cwd: workDir, timeout: 10_000 }, ); const cleanResult = await options.pi.exec("git", ["clean", "-fd", "--", "."], { cwd: workDir, timeout: 10_000 }); + const cleanIgnoredResult = await options.pi.exec("git", ["clean", "-fdX", "--", "."], { + cwd: workDir, + timeout: 10_000, + }); restoreAutoresearchFiles(preservedFiles); if (restoreResult.code !== 0) { return { @@ -578,6 +588,11 @@ async function revertFailedExperiment( error: `git clean failed: ${mergeStdoutStderr(cleanResult).trim() || `exit ${cleanResult.code}`}`, }; } + if (cleanIgnoredResult.code !== 0) { + return { + error: `git clean -X failed: ${mergeStdoutStderr(cleanIgnoredResult).trim() || `exit ${cleanIgnoredResult.code}`}`, + }; + } const dirtyCheckResult = await options.pi.exec( "git", ["status", "--porcelain=v1", "-z", "--untracked-files=all", "--", "."], diff --git a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts index a393c3647..a6281a29f 100644 --- a/packages/coding-agent/src/autoresearch/tools/run-experiment.ts +++ b/packages/coding-agent/src/autoresearch/tools/run-experiment.ts @@ -303,6 +303,10 @@ export function createRunExperimentTool( runDirectory, runNumber, }; + runtime.autoResumeArmed = true; + runtime.lastAutoResumePendingRunNumber = null; + options.dashboard.updateWidget(ctx, runtime); + options.dashboard.requestRender(); await Bun.write( runJsonPath, diff --git a/packages/coding-agent/src/autoresearch/types.ts b/packages/coding-agent/src/autoresearch/types.ts index 27ec82d02..e14fadac9 100644 --- a/packages/coding-agent/src/autoresearch/types.ts +++ b/packages/coding-agent/src/autoresearch/types.ts @@ -135,7 +135,9 @@ export interface RunningExperiment { export interface AutoresearchRuntime { autoresearchMode: boolean; + autoResumeArmed: boolean; dashboardExpanded: boolean; + lastAutoResumePendingRunNumber: number | null; lastRunChecks: ChecksResult | null; lastRunDuration: number | null; lastRunAsi: ASIData | null; diff --git a/packages/coding-agent/src/patch/hashline.ts b/packages/coding-agent/src/patch/hashline.ts index 922eed2b6..58330de73 100644 --- a/packages/coding-agent/src/patch/hashline.ts +++ b/packages/coding-agent/src/patch/hashline.ts @@ -567,7 +567,7 @@ export function applyHashlineEdits( const tag = formatLineTag(endLine + 1, nextSurvivingLine); warnings.push( `Possible boundary duplication: your last replacement line \`${trimmedLast}\` is identical to the next surviving line ${tag}. ` + - `If you meant to replace the entire block, set \`end\` to ${tag} instead.`, + `If you meant to replace the entire block, set \`end\` to ${tag} instead.`, ); } } diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index ff4f3702c..61749db87 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -114,4 +114,4 @@ When adding a sibling declaration, prefer `prepend` on the next declaration. - For a block, either replace only the body or replace the whole block. Do not split block boundaries. - `content` must be literal file content with matching indentation. If the file uses tabs, use real tabs. - Do not use this tool to reformat or clean up unrelated code. - + \ No newline at end of file diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 42ccc162a..3c2011b06 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -6,8 +6,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { Agent, AgentBusyError, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; -import { type AssistantMessage, getBundledModel, type ToolCall } from "@oh-my-pi/pi-ai"; +import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { type AssistantMessage, getBundledModel, type Message, type ToolCall } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -15,6 +15,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { Snowflake } from "@oh-my-pi/pi-utils"; import { Type } from "@sinclair/typebox"; @@ -112,6 +113,16 @@ describe("AgentSession concurrent prompt guard", () => { return session; } + async function waitFor(predicate: () => boolean, timeoutMs = 500): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (predicate()) return; + await Bun.sleep(10); + } + + throw new Error("Timed out waiting for condition"); + } + it("should throw when prompt() called while streaming", async () => { await createSession(); @@ -167,7 +178,7 @@ describe("AgentSession concurrent prompt guard", () => { it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let firstStream: MockAssistantStream | undefined; - const callMessages: AgentMessage[][] = []; + const callMessages: Message[][] = []; const agent = new Agent({ getApiKey: () => "test-key", @@ -176,6 +187,7 @@ describe("AgentSession concurrent prompt guard", () => { systemPrompt: "Test", tools: [], }, + convertToLlm, streamFn: (_model, context) => { callMessages.push([...context.messages]); const stream = new MockAssistantStream(); @@ -206,7 +218,7 @@ describe("AgentSession concurrent prompt guard", () => { }); const firstPrompt = session.prompt("First message"); - await Bun.sleep(10); + await waitFor(() => session.isStreaming && firstStream !== undefined && callMessages.length === 1); await session.sendCustomMessage( { @@ -227,10 +239,15 @@ describe("AgentSession concurrent prompt guard", () => { expect(callMessages).toHaveLength(2); expect( - callMessages[1]?.some( - message => - message.role === "custom" && "customType" in message && message.customType === "autoresearch-resume", - ), + callMessages[1]?.some(message => { + if (typeof message.content === "string") { + return message.content.includes("Hidden stop reaction"); + } + + return message.content.some( + content => content.type === "text" && content.text.includes("Hidden stop reaction"), + ); + }), ).toBe(true); }); diff --git a/packages/coding-agent/test/autoresearch-state.test.ts b/packages/coding-agent/test/autoresearch-state.test.ts index a5d20eb17..db59e877a 100644 --- a/packages/coding-agent/test/autoresearch-state.test.ts +++ b/packages/coding-agent/test/autoresearch-state.test.ts @@ -297,6 +297,12 @@ describe("autoresearch command guard", () => { expect(isAutoresearchShCommand("echo hi; autoresearch.sh")).toBe(false); expect(isAutoresearchShCommand("bash -lc 'autoresearch.sh'")).toBe(false); }); + + it("rejects chained or redirected benchmark commands even when autoresearch.sh comes first", () => { + expect(isAutoresearchShCommand("bash autoresearch.sh && touch /tmp/marker")).toBe(false); + expect(isAutoresearchShCommand("./autoresearch.sh | tee run.log")).toBe(false); + expect(isAutoresearchShCommand("./autoresearch.sh > run.log")).toBe(false); + }); }); interface AutoresearchCommandHarness { @@ -394,6 +400,9 @@ function createAutoresearchCommandHarness( } interface AutoresearchLifecycleHarness { + beforeAgentStartHandler: + | ((event: { systemPrompt: string }, ctx: ExtensionContext) => Promise | unknown) + | undefined; sessionStartHandler: ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) | undefined; sessionSwitchHandler: ((event: SessionSwitchEvent, ctx: ExtensionContext) => Promise | void) | undefined; agentEndHandler: ((event: unknown, ctx: ExtensionContext) => Promise | void) | undefined; @@ -475,6 +484,9 @@ function createAutoresearchLifecycleHarness(options: { } as unknown as ExtensionContext; return { + beforeAgentStartHandler: handlers.get("before_agent_start") as + | ((event: { systemPrompt: string }, ctx: ExtensionContext) => Promise | unknown) + | undefined, sessionStartHandler: handlers.get("session_start") as | ((event: SessionStartEvent, ctx: ExtensionContext) => Promise | void) | undefined, @@ -515,6 +527,7 @@ describe("autoresearch command startup", () => { "runtime_ms", "ms", "lower", + "memory_mb, rss_mb", "packages/coding-agent/src/autoresearch, packages/coding-agent/test", "packages/coding-agent/src/generated", "preserve output format", @@ -547,6 +560,7 @@ describe("autoresearch command startup", () => { { title: "Primary Metric Name", placeholder: "runtime_ms" }, { title: "Metric Unit", placeholder: "ms" }, { title: "Metric Direction", placeholder: "lower" }, + { title: "Tradeoff Metrics", placeholder: "" }, { title: "Files in Scope", placeholder: "packages/coding-agent/src/autoresearch" }, { title: "Off Limits", placeholder: "" }, { title: "Constraints", placeholder: "" }, @@ -558,6 +572,8 @@ describe("autoresearch command startup", () => { expect(harness.sentMessages[0]).toContain("primary metric: `runtime_ms`"); expect(harness.sentMessages[0]).toContain("metric unit: `ms`"); expect(harness.sentMessages[0]).toContain("direction: `lower`"); + expect(harness.sentMessages[0]).toContain("`memory_mb`"); + expect(harness.sentMessages[0]).toContain("`rss_mb`"); expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/autoresearch`"); expect(harness.sentMessages[0]).toContain("`packages/coding-agent/src/generated`"); expect(harness.sentMessages[0]).toContain("preserve output format"); @@ -587,21 +603,17 @@ describe("autoresearch command startup", () => { await harness.command.handler("", harness.ctx); expect(harness.inputCalls).toEqual([]); - expect(harness.sentMessages).toEqual([ - [ - "Resume autoresearch from the attached notes.", - "", - `@${autoresearchMdPath}`, - "", - "Using dedicated git branch `autoresearch/existing-20260322`.", - "", - "Use the notes as the source of truth for the current direction, scope, and constraints.", - "- inspect recent git history for context", - "- inspect `autoresearch.jsonl` if it exists", - "- continue the most promising unfinished branch", - "- keep iterating until interrupted or until the configured iteration cap is reached", - ].join("\n"), - ]); + expect(harness.sentMessages).toHaveLength(1); + expect(harness.sentMessages[0]).toContain("Resume autoresearch from the attached notes."); + expect(harness.sentMessages[0]).toContain(`@${autoresearchMdPath}`); + expect(harness.sentMessages[0]).toContain("Using dedicated git branch `autoresearch/existing-20260322`."); + expect(harness.sentMessages[0]).toContain( + "Use the notes as the source of truth for the current direction, scope, and constraints.", + ); + expect(harness.sentMessages[0]).toContain("- inspect `autoresearch.jsonl` if it exists"); + expect(harness.sentMessages[0]).toContain( + "- continue the most promising unfinished direction on the current protected branch", + ); }); it("includes explicit resume context when the user resumes with additional instructions", async () => { @@ -642,6 +654,7 @@ describe("autoresearch command startup", () => { "runtime_ms", "ms", "lower", + "", "packages/coding-agent/src/autoresearch", "", "", @@ -741,12 +754,16 @@ describe("autoresearch command startup", () => { "runtime_ms", "ms", "lower", + "", "packages/coding-agent/src/autoresearch", "", "", ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse" && args[1] === "--show-prefix") { + return { code: 0, stderr: "", stdout: "" }; + } if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; if (args[0] === "branch" && args[1] === "--show-current") { return { code: 0, stderr: "", stdout: "main\n" }; @@ -782,12 +799,16 @@ describe("autoresearch command startup", () => { "runtime_ms", "ms", "lower", + "", "packages/coding-agent/src/autoresearch", "", "", ], async (command, args) => { if (command !== "git") return { code: 1, stderr: "unexpected command", stdout: "" }; + if (args[0] === "rev-parse" && args[1] === "--show-prefix") { + return { code: 0, stderr: "", stdout: "" }; + } if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` }; if (args[0] === "branch" && args[1] === "--show-current") { return { code: 0, stderr: "", stdout: "main\n" }; @@ -814,6 +835,7 @@ describe("autoresearch command startup", () => { "runtime_ms", "ms", "lower", + "", "packages/coding-agent/src/autoresearch", "", "", @@ -1057,6 +1079,99 @@ describe("autoresearch auto-resume", () => { triggerTurn: true, }); }); + + it("does not enqueue another hidden turn after a passive autoresearch turn with no pending run", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + `${JSON.stringify({ type: "config", metricName: "runtime_ms", scopePaths: ["src"] })}\n`, + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["init_experiment", "run_experiment", "log_experiment"], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + await harness.agentEndHandler?.({}, harness.ctx); + + expect(harness.sentMessages).toEqual([]); + }); + + it("renders the high-signal prompt sections for playbooks, backlog, recent runs, and pending runs", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + fs.writeFileSync(path.join(dir, "autoresearch.md"), "# Autoresearch\n"); + fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Local Playbook\n"); + fs.writeFileSync(path.join(dir, "autoresearch.ideas.md"), "- try batching\n"); + fs.writeFileSync(path.join(dir, "autoresearch.checks.sh"), "#!/usr/bin/env bash\n"); + fs.writeFileSync( + path.join(dir, "autoresearch.jsonl"), + [ + JSON.stringify({ + type: "config", + metricName: "runtime_ms", + metricUnit: "ms", + scopePaths: ["src"], + }), + JSON.stringify({ + run: 1, + commit: "aaaaaaa", + metric: 10, + status: "keep", + description: "baseline", + timestamp: 1, + asi: { hypothesis: "baseline" }, + }), + JSON.stringify({ + run: 2, + commit: "bbbbbbb", + metric: 9, + status: "discard", + description: "too noisy", + timestamp: 2, + asi: { + hypothesis: "raise cache size", + rollback_reason: "noise", + next_action_hint: "re-test with more samples", + }, + }), + ].join("\n"), + ); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0003", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedPrimary: 8, + runNumber: 3, + }), + ); + + const harness = createAutoresearchLifecycleHarness({ + activeTools: ["init_experiment", "run_experiment", "log_experiment"], + controlEntries: [{ type: "custom", customType: "autoresearch-control", data: { mode: "on", goal: "x" } }], + cwd: dir, + }); + + await harness.sessionStartHandler?.({ type: "session_start" } as SessionStartEvent, harness.ctx); + const result = await harness.beforeAgentStartHandler?.({ systemPrompt: "BASE" }, harness.ctx); + const systemPrompt = + typeof result === "object" && result !== null && "systemPrompt" in result + ? String((result as { systemPrompt: string }).systemPrompt) + : ""; + + expect(systemPrompt).toContain("### Local Playbook"); + expect(systemPrompt).toContain("### Current Segment Snapshot"); + expect(systemPrompt).toContain("### Pending Run"); + expect(systemPrompt).toContain("### Ideas backlog"); + expect(systemPrompt).toContain("Recent runs:"); + expect(systemPrompt).toContain("finish the `log_experiment` step before starting another benchmark"); + }); }); describe("autoresearch lifecycle tool activation", () => { diff --git a/packages/coding-agent/test/autoresearch-tools.test.ts b/packages/coding-agent/test/autoresearch-tools.test.ts index 068c16452..19ae7f6f1 100644 --- a/packages/coding-agent/test/autoresearch-tools.test.ts +++ b/packages/coding-agent/test/autoresearch-tools.test.ts @@ -245,6 +245,24 @@ describe("autoresearch tools", () => { }); const runtime = createSessionRuntime(); + runtime.lastRunChecks = { pass: true, output: "stale", duration: 1 }; + runtime.lastRunDuration = 1; + runtime.lastRunAsi = { hypothesis: "stale" }; + runtime.lastRunArtifactDir = path.join(dir, ".autoresearch", "runs", "9999"); + runtime.lastRunNumber = 99; + runtime.lastRunSummary = { + checksDurationSeconds: 1, + checksPass: true, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: { hypothesis: "stale" }, + parsedMetrics: { runtime_ms: 10 }, + parsedPrimary: 10, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "9999"), + runNumber: 99, + }; const tool = createInitExperimentTool({ dashboard: createDashboardStub(), getRuntime: () => runtime, @@ -291,6 +309,12 @@ describe("autoresearch tools", () => { expect(configEntry.offLimits).toEqual(["src/generated"]); expect(configEntry.constraints).toEqual(["keep behavior stable"]); expect(configEntry.segmentFingerprint).toBe(createFingerprint(dir)); + expect(runtime.lastRunChecks).toBeNull(); + expect(runtime.lastRunDuration).toBeNull(); + expect(runtime.lastRunAsi).toBeNull(); + expect(runtime.lastRunArtifactDir).toBeNull(); + expect(runtime.lastRunNumber).toBeNull(); + expect(runtime.lastRunSummary).toBeNull(); }); it("rejects init_experiment when the passed contract no longer matches autoresearch.md", async () => { @@ -349,6 +373,53 @@ describe("autoresearch tools", () => { expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false); }); + it("rejects init_experiment while a previous run is still pending", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedPrimary: 10, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + const tool = createInitExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + + const result = await tool.execute( + "init-pending", + { + name: "Blocked", + metric_name: "runtime_ms", + metric_unit: "ms", + direction: "lower", + benchmark_command: "bash autoresearch.sh", + scope_paths: ["src"], + off_limits: [], + constraints: [], + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("has not been logged yet"), + }); + expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false); + }); + it("refuses to start a new benchmark while a previous run artifact is still unlogged", async () => { const dir = makeTempDir(); tempDirs.push(dir); @@ -512,6 +583,8 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-force-secondary-accept`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-keep`.cwd(dir).quiet(); fs.writeFileSync(path.join(dir, "src", "in-scope.ts"), "export const value = 3;\n"); fs.writeFileSync(path.join(dir, "autoresearch.program.md"), "# Strategy\n\n- focus on in-scope edits\n"); @@ -618,6 +691,8 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-max-iterations-accept`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-force-secondary`.cwd(dir).quiet(); fs.writeFileSync(path.join(dir, "src", "out-of-scope.ts"), "export const value = 99;\n"); await Bun.write( @@ -707,6 +782,8 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-discard-cleanup`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-max-iterations`.cwd(dir).quiet(); fs.writeFileSync(path.join(dir, "src", "generated", "index.ts"), "export const value = 2;\n"); await Bun.write( @@ -776,6 +853,7 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-discard`.cwd(dir).quiet(); await Bun.write( path.join(dir, ".autoresearch", "runs", "0003", "run.json"), @@ -861,6 +939,280 @@ describe("autoresearch tools", () => { expect(runtime.state.results).toHaveLength(2); }); + it("rejects log_experiment when configured secondary metrics are missing", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { memory_mb: 32, runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.secondaryMetrics = [ + { name: "memory_mb", unit: "mb" }, + { name: "tokens", unit: "" }, + ]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { memory_mb: 32, runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-missing-secondary", + { + commit: "initial", + metric: 9, + status: "discard", + description: "missing tokens metric", + metrics: { memory_mb: 32 }, + asi: { + hypothesis: "watch memory only", + rollback_reason: "missing required metrics", + next_action_hint: "include all configured tradeoff metrics", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("missing secondary metrics: tokens"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("rejects new secondary metrics unless force is enabled", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.secondaryMetrics = [{ name: "memory_mb", unit: "mb" }]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-new-secondary", + { + commit: "initial", + metric: 9, + status: "discard", + description: "introduce tokens metric", + metrics: { memory_mb: 32, tokens: 100 }, + asi: { + hypothesis: "watch an extra tradeoff", + rollback_reason: "needs explicit opt-in", + next_action_hint: "retry with force if the metric matters", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("new secondary metrics require force=true: tokens"), + }); + expect(runtime.state.results).toHaveLength(0); + }); + + it("accepts a new secondary metric when force is enabled", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + + await $`git init`.cwd(dir).quiet(); + await $`git config user.email test@example.com`.cwd(dir).quiet(); + await $`git config user.name Test User`.cwd(dir).quiet(); + await $`git add .`.cwd(dir).quiet(); + await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-force-secondary-accept`.cwd(dir).quiet(); + + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.secondaryMetrics = [{ name: "memory_mb", unit: "mb" }]; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: createGitApi(), + }); + const result = await tool.execute( + "call-force-secondary", + { + commit: "initial", + metric: 9, + status: "discard", + description: "force extra metric", + force: true, + metrics: { memory_mb: 32, tokens: 100 }, + asi: { + hypothesis: "capture an extra tradeoff", + rollback_reason: "benchmark was flat", + next_action_hint: "keep collecting tokens when useful", + }, + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("Logged run #1: discard"), + }); + expect(runtime.state.secondaryMetrics).toContainEqual({ name: "tokens", unit: "" }); + }); + + it("rejects log_experiment at the tool boundary when asi is missing", async () => { + const dir = makeTempDir(); + tempDirs.push(dir); + writeAutoresearchWorkspace(dir); + await Bun.write( + path.join(dir, ".autoresearch", "runs", "0001", "run.json"), + JSON.stringify({ + command: "bash autoresearch.sh", + completedAt: new Date().toISOString(), + durationSeconds: 1, + exitCode: 0, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + runNumber: 1, + }), + ); + + const runtime = createSessionRuntime(); + runtime.state.metricName = "runtime_ms"; + runtime.state.metricUnit = "ms"; + runtime.state.segmentFingerprint = createFingerprint(dir); + runtime.lastRunSummary = { + checksDurationSeconds: 0, + checksPass: null, + checksTimedOut: false, + command: "bash autoresearch.sh", + durationSeconds: 1, + parsedAsi: null, + parsedMetrics: { runtime_ms: 9 }, + parsedPrimary: 9, + passed: true, + runDirectory: path.join(dir, ".autoresearch", "runs", "0001"), + runNumber: 1, + }; + + const tool = createLogExperimentTool({ + dashboard: createDashboardStub(), + getRuntime: () => runtime, + pi: {} as ExtensionAPI, + }); + const result = await tool.execute( + "call-missing-asi", + { + commit: "initial", + metric: 9, + status: "keep", + description: "missing asi", + }, + undefined, + undefined, + createContext(dir), + ); + + expect(result.content[0]).toEqual({ + type: "text", + text: expect.stringContaining("asi is required"), + }); + expect(runtime.state.results).toHaveLength(0); + expect(fs.existsSync(path.join(dir, "autoresearch.jsonl"))).toBe(false); + }); + it("requires failed benchmarks to be logged as crash", async () => { const dir = makeTempDir(); tempDirs.push(dir); @@ -1002,6 +1354,7 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-max-iterations-accept`.cwd(dir).quiet(); await Bun.write( path.join(dir, ".autoresearch", "runs", "0001", "run.json"), @@ -1164,6 +1517,7 @@ describe("autoresearch tools", () => { await $`git config user.name Test User`.cwd(dir).quiet(); await $`git add .`.cwd(dir).quiet(); await $`git commit -m initial`.cwd(dir).quiet(); + await $`git checkout -b autoresearch/test-discard-cleanup`.cwd(dir).quiet(); fs.mkdirSync(path.join(dir, "tmp-artifact"), { recursive: true }); fs.writeFileSync(path.join(dir, "tmp-artifact", "result.txt"), "temporary benchmark output\n"); diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index d1edbccd3..311ed743d 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -535,7 +535,9 @@ describe("applyHashlineEdits — heuristics", () => { ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("if (ok) {\n runSafe();\n}\n}\nafter();"); - expect(result.warnings).toBeUndefined(); + expect(result.warnings).toHaveLength(1); + expect(result.warnings?.[0]).toContain("Possible boundary duplication"); + expect(result.warnings?.[0]).toContain("set `end` to 3#RZ"); }); it("preserves duplicated trailing content when replacement re-emits the next line", () => { @@ -550,7 +552,9 @@ describe("applyHashlineEdits — heuristics", () => { ]; const result = applyHashlineEdits(content, edits); expect(result.lines).toBe("start\n newCall();\nnextCall();\nnextCall();\nafter();"); - expect(result.warnings).toBeUndefined(); + expect(result.warnings).toHaveLength(1); + expect(result.warnings?.[0]).toContain("Possible boundary duplication"); + expect(result.warnings?.[0]).toContain("set `end` to 3#HR"); }); it("preserves duplicated leading content when replacement re-emits the previous line", () => { From 4adaee02dabfaeb2a6b7dea1e83e6387c13e8285 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 23 Mar 2026 05:58:44 +0100 Subject: [PATCH 20/22] chore: bump version to 13.15.0 --- Cargo.lock | 6 ++--- Cargo.toml | 2 +- bun.lock | 34 +++++++++++++-------------- packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 ++ packages/ai/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 ++ packages/coding-agent/package.json | 2 +- packages/natives/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 ++ packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- 14 files changed, 35 insertions(+), 29 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 6301a7c08..c28da3648 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1245,9 +1245,9 @@ dependencies = [ [[package]] name = "html-to-markdown-rs" -version = "2.28.6" +version = "2.29.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6869b5e058b5ebb8c176269406b692d0695b4b19c36e532b56a2c355590978ae" +checksum = "9013679b8c3600142e5a8f742748c3c38c49d9fc50675dad62f8f1721090a85a" dependencies = [ "ahash", "astral-tl", @@ -2114,7 +2114,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "13.14.2" +version = "13.15.0" dependencies = [ "arboard", "ast-grep-core", diff --git a/Cargo.toml b/Cargo.toml index e2404e472..61fabde30 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "13.14.2" +version = "13.15.0" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index 4ec2cc838..14bb9820e 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "13.14.2", + "version": "13.15.0", "dependencies": { "@oh-my-pi/pi-ai": "workspace:*", "@oh-my-pi/pi-utils": "workspace:*", @@ -27,7 +27,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "13.14.2", + "version": "13.15.0", "bin": { "pi-ai": "./src/cli.ts", }, @@ -51,7 +51,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "13.14.2", + "version": "13.15.0", "bin": { "omp": "src/cli.ts", }, @@ -80,7 +80,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "13.14.2", + "version": "13.15.0", "dependencies": { "@oh-my-pi/pi-utils": "workspace:*", }, @@ -114,7 +114,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "13.14.2", + "version": "13.15.0", "bin": { "omp-stats": "./src/index.ts", }, @@ -139,7 +139,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "13.14.2", + "version": "13.15.0", "bin": { "omp-swarm": "src/cli.ts", }, @@ -152,7 +152,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "13.14.2", + "version": "13.15.0", "dependencies": { "@oh-my-pi/pi-natives": "workspace:*", "@oh-my-pi/pi-utils": "workspace:*", @@ -165,7 +165,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "13.14.2", + "version": "13.15.0", "dependencies": { "beautiful-mermaid": "^1.1", "winston": "^3.19", @@ -467,21 +467,21 @@ "@types/yauzl": ["@types/yauzl@2.10.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q=="], - "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260321.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260321.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260321.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260321.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260321.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260321.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260321.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260321.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-uScJZRWRxyi1l4EWwOtuO88Gh8sUTi0itcI4oKlyNtXkqik4Y7EHfs1sfYPDuAEJO3cvW6bqohHjGx3mcXSZzQ=="], + "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260322.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260322.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-CmzQTKvesYHmz3g92G+XPDis25ocvHqa/gK8m98w+bML99KJLEWQKVlvkLrYA85JiJEK+XBIiz+6lCgUqRkWXA=="], - "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260321.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-3LQP363bDCF/pmXqzhSCSkKXr1PpNl2elC167YFRPKRyJdrETiIwj3YAB8A6esn9D30pas5VLzfmeK/tUOf+6g=="], + "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260322.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-5wSilxwLGX5fMKJgsUkCBwOfW9GMG3WF5j77CVBOdFI7miFaR3JQaPzTA+uyHDMNIIeSDo1KtV77GT48Y/d0Xg=="], - "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260321.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-gCoKiv415CROgl0K8hEV8Lw/zvbYriWWmD7VxvpiQiTRqQmHppVXhLtb2OrGaPcsqpoBdYeCJQHN4wnohAkNLA=="], + "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260322.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-G806SrfxkYNAgZ9Xk53+OvbmIg9iD5hjaiD2QhDQL2aZjzy10D4MhcdaZEOoMfw0OI/PoJPYOiPD+9/x2kw3Lg=="], - "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260321.1", "", { "os": "linux", "cpu": "arm" }, "sha512-QuAFR9eFQzuqtKTIaJ5XkNR4i5Q55b1SE7fUcIAS528aY9j+5P1cMpvJa8aOBCuRKxfMgV5UtamolZKGWWzaMw=="], + "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "arm" }, "sha512-0a12pp19ELiNHMqTglfQQQNMsxvtzpjAa4qf12oMJoGyy+UnguKEmaaaCHdp75KvBXGDzlssfDAdiy+NirN19A=="], - "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260321.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-15z7UWt0PG870ktcUbaa0NogAjXIYT4pSFWlsc95u8+1aITrBTMQgqRih5qUH8bHke3eeYwbpjfXaU4gNmexvw=="], + "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-+FyomEEt3K8TBO//n3Ijr61SDM2F7cxZCVqGt+Wk3rLcOCQ2i+8+p64gdsZCmImy3CyP0hBnxPydEbyNkZLtvg=="], - "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260321.1", "", { "os": "linux", "cpu": "x64" }, "sha512-8yuzwkxQnNSpXjXK43Y5Pn6rBfNbJVIcd3Qh9n3Tzhgtr+lcoGgwgMvn8axnqaazkxIUB3PZuiGRcqr6XIq3LA=="], + "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "x64" }, "sha512-MviQe5x4WqQGv/Vhu4hcv2A0qTW/BTaZPbOLYCtvhuovNFO6D++ZmJAbHvA0h/bJEaNTgxKZdZPHMpCfSEKfjA=="], - "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260321.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-fCUk/VElUjMFmE6iFAtsy5r7kLxeLggEHOTWuR0HGYIUQze6EyAdDFqMPFFxvbzpUyFQFpRfUa0I/Fa5tqKh8g=="], + "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260322.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-ibnMaXDJPSgMXKC61NHiFlww/xjAEINgc1mcn2ntTfuGHwduU4P9Bi038TxXg95Wmu3v6xIPIorXXsBOdE+p3Q=="], - "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260321.1", "", { "os": "win32", "cpu": "x64" }, "sha512-CWGyck7+sbNwOhcL+ObHhtKZe2/+Y6OZlEdWX2mHjpv8ef7ohUbPCdS94p+e7jbVY56w8NAce2Xx7ppn/C1Ucg=="], + "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260322.1", "", { "os": "win32", "cpu": "x64" }, "sha512-O+r1RToWBbGkK7NXC7DpraLObSWyxvSqRiSfr/BlZ351Cdq1q3121zCGzVtqERGeRtVoEMRrzS5ITOd6On/pCw=="], "@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, "sha512-PJFXFS4ZJKiJ9Qiuix6Dz/OwEIqHD7Dme1UwZhTK11vR+5dqW2ACbdndWQexBzCx+CPuMe5WBYQWCsFyGlQLlQ=="], @@ -941,7 +941,7 @@ "wrappy": ["wrappy@1.0.2", "", {}, "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ=="], - "ws": ["ws@8.19.0", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-blAT2mjOEIi0ZzruJfIhb3nps74PRWTCz1IjglWEEpQl5XS/UNama6u2/rjFkDDouqr4L67ry+1aGIALViWjDg=="], + "ws": ["ws@8.20.0", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-sAt8BhgNbzCtgGbt2OxmpuryO63ZoDk/sqaB/znQm94T4fCEsy/yV+7CdC1kJhOU9lboAEU7R3kquuycDoibVA=="], "y18n": ["y18n@5.0.8", "", {}, "sha512-0pfFzegeDWJHJIAmTLRP2DwHjdF5s7jo9tuztdQxAhINCdvS+3nGINqPd00AphqJR/0LhANUS6/+7SCb98YOfA=="], diff --git a/packages/agent/package.json b/packages/agent/package.json index a71b0b306..69ac86498 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "13.14.2", + "version": "13.15.0", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 531a6324e..bf0471f8a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [13.15.0] - 2026-03-23 + ### Added - Added `isUsageLimitError()` to `rate-limit-utils` as a single source of truth for detecting usage/quota limit errors across all providers diff --git a/packages/ai/package.json b/packages/ai/package.json index 4b76c9ed4..3b7279ee8 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "13.14.2", + "version": "13.15.0", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b55c0720e..43a9b92d2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,8 @@ # Changelog ## [Unreleased] + +## [13.15.0] - 2026-03-23 ### Breaking Changes - Changed hashline edit schema from flat `op`/`pos`/`end`/`lines` fields to structured `loc`/`content` format with location-specific objects diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 8c01dd0c1..df4387af1 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "13.14.2", + "version": "13.15.0", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/natives/package.json b/packages/natives/package.json index 406c2b00d..f49825a1c 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-natives", - "version": "13.14.2", + "version": "13.15.0", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index 90d6ee2bb..92c2e70c4 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "13.14.2", + "version": "13.15.0", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 3fc9f11f0..d8957689d 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "13.14.2", + "version": "13.15.0", "description": "Swarm orchestration extension for omp", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 2f717af51..b9e4df4f3 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [13.15.0] - 2026-03-23 + ### Added - Added `renderInlineMarkdown()` function to render inline markdown (bold, italic, code, links, strikethrough) to styled strings diff --git a/packages/tui/package.json b/packages/tui/package.json index 4e5ece4b8..40cadc9cd 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "13.14.2", + "version": "13.15.0", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 6008ea643..ced312d9e 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "13.14.2", + "version": "13.15.0", "description": "Shared utilities for pi packages", "homepage": "https://github.com/can1357/oh-my-pi", "author": "Can Boluk", From 870232104c66459b7b1eb41eaf09db88d35a0790 Mon Sep 17 00:00:00 2001 From: zamorakpds Date: Wed, 25 Mar 2026 12:40:08 +0100 Subject: [PATCH 21/22] Fix/resume timestamp mutation (#528) * fix(coding-agent): kept resumed sessions from reordering * fix(coding-agent): corrected resume state restoration --- packages/coding-agent/src/sdk.ts | 18 ++- .../coding-agent/src/session/agent-session.ts | 32 ++-- .../src/session/session-manager.ts | 13 +- .../test/agent-session-mcp-discovery.test.ts | 59 +++++-- .../test/sdk-mcp-discovery.test.ts | 101 +++++++++++- .../session-manager/file-operations.test.ts | 152 ++++++++++++++++++ 6 files changed, 332 insertions(+), 43 deletions(-) diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index c2cf91860..1d8e121f2 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -688,8 +688,10 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Check if session has existing data to restore const existingSession = logger.time("loadSession", () => sessionManager.buildSessionContext()); - const hasExistingSession = existingSession.messages.length > 0; - const hasThinkingEntry = sessionManager.getBranch().some(entry => entry.type === "thinking_level_change"); + const existingBranch = sessionManager.getBranch(); + const hasExistingSession = existingBranch.length > 0; + const hasThinkingEntry = existingBranch.some(entry => entry.type === "thinking_level_change"); + const hasServiceTierEntry = existingBranch.some(entry => entry.type === "service_tier_change"); const hasExplicitModel = options.model !== undefined || options.modelPattern !== undefined; const modelMatchPreferences = { @@ -1428,6 +1430,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} openaiWebsocketSetting === "on" ? true : openaiWebsocketSetting === "off" ? false : undefined; const serviceTierSetting = settings.get("serviceTier"); + const initialServiceTier = hasServiceTierEntry + ? existingSession.serviceTier + : serviceTierSetting === "none" + ? undefined + : serviceTierSetting; + agent = new Agent({ initialState: { systemPrompt, @@ -1449,7 +1457,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} minP: settings.get("minP") >= 0 ? settings.get("minP") : undefined, presencePenalty: settings.get("presencePenalty") >= 0 ? settings.get("presencePenalty") : undefined, repetitionPenalty: settings.get("repetitionPenalty") >= 0 ? settings.get("repetitionPenalty") : undefined, - serviceTier: serviceTierSetting === "none" ? undefined : serviceTierSetting, + serviceTier: initialServiceTier, kimiApiFormat: settings.get("providers.kimiApiFormat") ?? "anthropic", preferWebsockets: preferOpenAICodexWebsockets, getToolContext: tc => toolContextStore.getContext(tc), @@ -1487,9 +1495,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Restore messages if session has existing data if (hasExistingSession) { agent.replaceMessages(existingSession.messages); - if (!hasThinkingEntry) { - sessionManager.appendThinkingLevelChange(thinkingLevel); - } } else { // Save initial model and thinking level for new sessions so they can be restored on resume if (model) { @@ -1520,6 +1525,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} mcpDiscoveryEnabled, initialSelectedMCPToolNames, defaultSelectedMCPToolNames, + persistInitialMCPToolSelection: !hasExistingSession, defaultSelectedMCPServerNames: [...discoveryDefaultServers], ttsrManager, obfuscator, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index ffe123444..f6c2c980e 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -225,6 +225,8 @@ export interface AgentSessionConfig { mcpDiscoveryEnabled?: boolean; /** MCP tool names to activate for the current session when discovery mode is enabled. */ initialSelectedMCPToolNames?: string[]; + /** Whether constructor-provided MCP defaults should be persisted immediately. */ + persistInitialMCPToolSelection?: boolean; /** MCP server names whose tools should seed discovery-mode sessions whenever those servers are connected. */ defaultSelectedMCPServerNames?: string[]; /** MCP tool names that should seed brand-new sessions created from this AgentSession. */ @@ -485,8 +487,11 @@ export class AgentSession { this.#pruneSelectedMCPToolNames(); const persistedSelectedMCPToolNames = this.sessionManager.buildSessionContext().selectedMCPToolNames; const currentSelectedMCPToolNames = this.getSelectedMCPToolNames(); + const persistInitialMCPToolSelection = + config.persistInitialMCPToolSelection ?? this.sessionManager.getBranch().length === 0; if ( this.#mcpDiscoveryEnabled && + persistInitialMCPToolSelection && !this.#selectedMCPToolNamesMatch(persistedSelectedMCPToolNames, currentSelectedMCPToolNames) ) { this.sessionManager.appendMCPToolSelection(currentSelectedMCPToolNames); @@ -5093,21 +5098,18 @@ export class AgentSession { const hasThinkingEntry = this.sessionManager.getBranch().some(entry => entry.type === "thinking_level_change"); const hasServiceTierEntry = this.sessionManager.getBranch().some(entry => entry.type === "service_tier_change"); const defaultThinkingLevel = this.settings.get("defaultThinkingLevel"); - - if (hasThinkingEntry) { - this.setThinkingLevel(sessionContext.thinkingLevel as ThinkingLevel | undefined); - } else { - const effectiveDefaultThinkingLevel = resolveThinkingLevelForModel(this.model, defaultThinkingLevel); - this.#thinkingLevel = effectiveDefaultThinkingLevel; - this.agent.setThinkingLevel(toReasoningEffort(effectiveDefaultThinkingLevel)); - this.sessionManager.appendThinkingLevelChange(effectiveDefaultThinkingLevel); - } - - if (hasServiceTierEntry) { - this.agent.serviceTier = sessionContext.serviceTier; - } else { - this.sessionManager.appendServiceTierChange(this.serviceTier ?? null); - } + const configuredServiceTier = this.settings.get("serviceTier"); + const nextThinkingLevel = resolveThinkingLevelForModel( + this.model, + hasThinkingEntry ? (sessionContext.thinkingLevel as ThinkingLevel | undefined) : defaultThinkingLevel, + ); + this.#thinkingLevel = nextThinkingLevel; + this.agent.setThinkingLevel(toReasoningEffort(nextThinkingLevel)); + this.agent.serviceTier = hasServiceTierEntry + ? sessionContext.serviceTier + : configuredServiceTier === "none" + ? undefined + : configuredServiceTier; this.#reconnectToAgent(); return true; diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index 36302e806..35192598c 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -1379,6 +1379,7 @@ export class SessionManager { #sessionName: string | undefined; #sessionFile: string | undefined; #flushed: boolean = false; + #needsFullRewriteOnNextPersist: boolean = false; #fileEntries: FileEntry[] = []; #byId: Map = new Map(); #labelsById: Map = new Map(); @@ -1441,9 +1442,7 @@ export class SessionManager { this.#sessionId = header?.id ?? Snowflake.next(); this.#sessionName = header?.title; - if (migrateToCurrentVersion(this.#fileEntries)) { - await this.#rewriteFile(); - } + this.#needsFullRewriteOnNextPersist = migrateToCurrentVersion(this.#fileEntries); await resolveBlobRefsInEntries(this.#fileEntries, this.#blobStore); @@ -1630,6 +1629,7 @@ export class SessionManager { this.#labelsById.clear(); this.#leafId = null; this.#flushed = false; + this.#needsFullRewriteOnNextPersist = false; this.#usageStatistics = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, premiumRequests: 0, cost: 0 }; if (this.persist) { @@ -1772,6 +1772,7 @@ export class SessionManager { this.#fileEntries.map(entry => prepareEntryForPersistence(entry, this.#blobStore)), ); await this.#writeEntriesAtomically(entries); + this.#needsFullRewriteOnNextPersist = false; this.#flushed = true; }); } @@ -1786,7 +1787,7 @@ export class SessionManager { */ async ensureOnDisk(): Promise { if (!this.persist || !this.#sessionFile) return; - if (this.#flushed) return; + if (this.#flushed && !this.#needsFullRewriteOnNextPersist) return; await this.#rewriteFile(); } @@ -1919,12 +1920,12 @@ export class SessionManager { const hasAssistant = this.#fileEntries.some(e => e.type === "message" && e.message.role === "assistant"); if (!hasAssistant) { - // Mark as not flushed so when assistant arrives, all entries get written + // Mark as not flushed so when assistant arrives, all entries get written. this.#flushed = false; return; } - if (!this.#flushed) { + if (this.#needsFullRewriteOnNextPersist || !this.#flushed) { // Full flush: rewrite the entire file atomically to avoid // duplicating entries if the file already exists (e.g. from ensureOnDisk). void this.#rewriteFile(); diff --git a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts index 698f28fb6..2d3614d60 100644 --- a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts +++ b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts @@ -2,9 +2,8 @@ import { afterEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import type { AgentTool } from "@oh-my-pi/pi-agent-core"; -import { Agent } from "@oh-my-pi/pi-agent-core"; -import type { Model } from "@oh-my-pi/pi-ai"; +import { Agent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import { Effort, type Model } from "@oh-my-pi/pi-ai"; import { Type } from "@sinclair/typebox"; import { Settings } from "../src/config/settings"; import type { CustomTool } from "../src/extensibility/custom-tools/types"; @@ -398,7 +397,7 @@ describe("AgentSession MCP discovery", () => { expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual([]); }); - it("persists corrected empty MCP selections when restored tools are unavailable", async () => { + it("restores unavailable MCP selections in memory without rewriting the persisted session selection", async () => { const readTool = createBasicTool("read", "Read"); const sessionManager = SessionManager.inMemory(); sessionManager.appendMCPToolSelection(["mcp_docs_search"]); @@ -422,7 +421,7 @@ describe("AgentSession MCP discovery", () => { sessions.push(session); expect(session.getSelectedMCPToolNames()).toEqual([]); - expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual([]); + expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual(["mcp_docs_search"]); }); it("restores MCP discovery selections when branching to a context without them", async () => { @@ -562,6 +561,11 @@ describe("AgentSession MCP discovery", () => { content: "start", timestamp: Date.now(), }); + sessionManager.appendMessage({ + role: "user", + content: "follow up", + timestamp: Date.now(), + }); const toolRegistry = new Map([ [readTool.name, readTool], [docsSearchTool.name, docsSearchTool], @@ -595,7 +599,7 @@ describe("AgentSession MCP discovery", () => { expect(session.systemPrompt).toBe("tools:read,mcp_docs_search"); }); - it("does not leak MCP defaults across session switches without persisted selections", async () => { + it("restores session defaults in memory across session switches without rewriting sessions missing persisted metadata", async () => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-agent-session-mcp-switch-")); tempDirs.push(tempDir); const readTool = createBasicTool("read", "Read"); @@ -606,18 +610,31 @@ describe("AgentSession MCP discovery", () => { ]); const olderSessionManager = SessionManager.create(tempDir, tempDir); + olderSessionManager.appendMessage({ + role: "user", + content: "older session", + timestamp: Date.now(), + }); const olderSessionFile = olderSessionManager.getSessionFile(); expect(olderSessionFile).toBeString(); - await olderSessionManager.flush(); + await olderSessionManager.rewriteEntries(); + const olderSessionBeforeSwitch = fs.readFileSync(olderSessionFile!, "utf8"); + const olderSessionMtimeBeforeSwitch = fs.statSync(olderSessionFile!).mtimeMs; const sessionManager = SessionManager.create(tempDir, tempDir); const originalSessionFile = sessionManager.getSessionFile(); expect(originalSessionFile).toBeString(); await sessionManager.flush(); + const reasoningModel: Model<"openai-responses"> = { + ...createModel(), + reasoning: true, + thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.Medium }, + }; + const agent = new Agent({ initialState: { - model: createModel(), + model: reasoningModel, systemPrompt: "initial", tools: [readTool, docsSearchTool], messages: sessionManager.buildSessionContext().messages, @@ -626,7 +643,11 @@ describe("AgentSession MCP discovery", () => { const session = new AgentSession({ agent, sessionManager, - settings: Settings.isolated({ "mcp.discoveryMode": true }), + settings: Settings.isolated({ + "mcp.discoveryMode": true, + defaultThinkingLevel: "high", + serviceTier: "priority", + }), modelRegistry: {} as never, toolRegistry, mcpDiscoveryEnabled: true, @@ -637,17 +658,37 @@ describe("AgentSession MCP discovery", () => { sessions.push(session); expect(session.getSelectedMCPToolNames()).toEqual(["mcp_docs_search"]); + sessionManager.appendThinkingLevelChange(ThinkingLevel.High); + sessionManager.appendServiceTierChange("flex"); + sessionManager.appendMCPToolSelection(["mcp_docs_search"]); + expect(sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.High); + expect(sessionManager.buildSessionContext().serviceTier).toBe("flex"); + expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual(["mcp_docs_search"]); expect(sessionManager.buildSessionContext().hasPersistedMCPToolSelection).toBe(true); + await sessionManager.rewriteEntries(); + const originalSessionBeforeSwitch = fs.readFileSync(originalSessionFile!, "utf8"); + const originalSessionMtimeBeforeSwitch = fs.statSync(originalSessionFile!).mtimeMs; + await Bun.sleep(20); await session.switchSession(olderSessionFile!); + expect(session.sessionFile).toBe(olderSessionFile); + expect(session.thinkingLevel).toBe(ThinkingLevel.Medium); + expect(session.serviceTier).toBe("priority"); expect(session.getSelectedMCPToolNames()).toEqual([]); expect(session.getActiveToolNames()).toEqual(["read"]); expect(session.systemPrompt).toBe("tools:read"); + expect(fs.readFileSync(olderSessionFile!, "utf8")).toBe(olderSessionBeforeSwitch); + expect(fs.statSync(olderSessionFile!).mtimeMs).toBe(olderSessionMtimeBeforeSwitch); await session.switchSession(originalSessionFile!); + expect(session.sessionFile).toBe(originalSessionFile); + expect(session.thinkingLevel).toBe(ThinkingLevel.Medium); + expect(session.serviceTier).toBe("flex"); expect(session.getSelectedMCPToolNames()).toEqual(["mcp_docs_search"]); expect(session.getActiveToolNames()).toEqual(["read", "mcp_docs_search"]); expect(session.systemPrompt).toBe("tools:read,mcp_docs_search"); + expect(fs.readFileSync(originalSessionFile!, "utf8")).toBe(originalSessionBeforeSwitch); + expect(fs.statSync(originalSessionFile!).mtimeMs).toBe(originalSessionMtimeBeforeSwitch); }); it("restores explicit MCP defaults after startup outage once tools recover in a new session", async () => { diff --git a/packages/coding-agent/test/sdk-mcp-discovery.test.ts b/packages/coding-agent/test/sdk-mcp-discovery.test.ts index 75185f474..71b957579 100644 --- a/packages/coding-agent/test/sdk-mcp-discovery.test.ts +++ b/packages/coding-agent/test/sdk-mcp-discovery.test.ts @@ -2,7 +2,8 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { getBundledModel } from "@oh-my-pi/pi-ai"; +import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import { Effort, getBundledModel, type Model } from "@oh-my-pi/pi-ai"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { CustomTool } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools/types"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; @@ -24,6 +25,22 @@ function createMcpCustomTool(name: string, serverName: string, mcpToolName: stri } as CustomTool; } +function createReasoningModel(): Model<"openai-responses"> { + return { + id: "mock-reasoning", + name: "mock-reasoning", + api: "openai-responses", + provider: "openai", + baseUrl: "https://example.invalid", + reasoning: true, + thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.High }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8192, + maxTokens: 2048, + }; +} + describe("createAgentSession MCP discovery prompt gating", () => { let tempDir: string; @@ -149,14 +166,18 @@ describe("createAgentSession MCP discovery prompt gating", () => { expect(searchTool?.description).toContain("Total discoverable MCP tools loaded: 1."); expect(searchTool?.description).toContain("- `server_name`"); }); - it("restores discovered MCP tools when resuming a persisted session", async () => { + it("restores explicit MCP, thinking, and service-tier entries when resuming without rewriting the session file", async () => { const firstManager = SessionManager.create(tempDir, tempDir); const { session: firstSession } = await createAgentSession({ cwd: tempDir, agentDir: tempDir, sessionManager: firstManager, - settings: Settings.isolated({ "mcp.discoveryMode": true }), - model: getBundledModel("openai", "gpt-4o-mini"), + settings: Settings.isolated({ + "mcp.discoveryMode": true, + defaultThinkingLevel: "high", + serviceTier: "priority", + }), + model: createReasoningModel(), disableExtensionDiscovery: true, skills: [], contextFiles: [], @@ -171,19 +192,28 @@ describe("createAgentSession MCP discovery prompt gating", () => { ], }); await firstSession.activateDiscoveredMCPTools(["mcp_slack_post_message"]); + firstSession.sessionManager.appendThinkingLevelChange(ThinkingLevel.Off); + firstSession.sessionManager.appendServiceTierChange("priority"); + expect(firstSession.sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.Off); expect(firstSession.getSelectedMCPToolNames()).toEqual(["mcp_slack_post_message"]); const sessionFile = firstSession.sessionFile; expect(sessionFile).toBeDefined(); await firstSession.sessionManager.rewriteEntries(); + const persistedBeforeResume = fs.readFileSync(sessionFile!, "utf8"); + const persistedMtimeBeforeResume = fs.statSync(sessionFile!).mtimeMs; + await Bun.sleep(20); await firstSession.dispose(); - const resumedManager = await SessionManager.open(sessionFile!, tempDir); const { session: resumedSession } = await createAgentSession({ cwd: tempDir, agentDir: tempDir, sessionManager: resumedManager, - settings: Settings.isolated({ "mcp.discoveryMode": true }), - model: getBundledModel("openai", "gpt-4o-mini"), + settings: Settings.isolated({ + "mcp.discoveryMode": true, + defaultThinkingLevel: "high", + serviceTier: "none", + }), + model: createReasoningModel(), disableExtensionDiscovery: true, skills: [], contextFiles: [], @@ -198,16 +228,73 @@ describe("createAgentSession MCP discovery prompt gating", () => { ], }); try { + expect(resumedSession.thinkingLevel).toBe(ThinkingLevel.Off); + expect(resumedSession.serviceTier).toBe("priority"); expect(resumedSession.getSelectedMCPToolNames()).toEqual(["mcp_slack_post_message"]); expect(resumedSession.getActiveToolNames()).toEqual( expect.arrayContaining(["read", "search_tool_bm25", "mcp_slack_post_message"]), ); expect(resumedSession.systemPrompt).toContain("mcp_slack_post_message"); + expect(fs.readFileSync(sessionFile!, "utf8")).toBe(persistedBeforeResume); + expect(fs.statSync(sessionFile!).mtimeMs).toBe(persistedMtimeBeforeResume); } finally { await resumedSession.dispose(); } }); + it("restores fallback MCP, thinking, and service-tier state in memory without rewriting the session file", async () => { + const sessionManager = SessionManager.create(tempDir, tempDir); + sessionManager.appendMessage({ + role: "user", + content: "resume me", + timestamp: Date.now(), + }); + const sessionFile = sessionManager.getSessionFile(); + expect(sessionFile).toBeDefined(); + await sessionManager.rewriteEntries(); + const persistedBeforeResume = fs.readFileSync(sessionFile!, "utf8"); + const persistedMtimeBeforeResume = fs.statSync(sessionFile!).mtimeMs; + await Bun.sleep(20); + const resumedManager = await SessionManager.open(sessionFile!, tempDir); + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + sessionManager: resumedManager, + settings: Settings.isolated({ + "mcp.discoveryMode": true, + "mcp.discoveryDefaultServers": ["github"], + defaultThinkingLevel: "high", + serviceTier: "priority", + }), + model: createReasoningModel(), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + toolNames: ["read", "search_tool_bm25"], + customTools: [ + createMcpCustomTool("mcp_github_create_issue", "github", "create_issue"), + createMcpCustomTool("mcp_slack_post_message", "slack", "post_message"), + ], + }); + try { + expect(session.thinkingLevel).toBe(ThinkingLevel.High); + expect(session.serviceTier).toBe("priority"); + expect(session.getSelectedMCPToolNames()).toEqual(["mcp_github_create_issue"]); + expect(session.getActiveToolNames()).toEqual( + expect.arrayContaining(["read", "search_tool_bm25", "mcp_github_create_issue"]), + ); + expect(session.sessionManager.buildSessionContext().hasPersistedMCPToolSelection).toBe(false); + expect(fs.readFileSync(sessionFile!, "utf8")).toBe(persistedBeforeResume); + expect(fs.statSync(sessionFile!).mtimeMs).toBe(persistedMtimeBeforeResume); + } finally { + await session.dispose(); + } + }); + it("keeps a cleared MCP selection empty when resuming with explicitly requested MCP tools", async () => { const firstManager = SessionManager.create(tempDir, tempDir); const { session: firstSession } = await createAgentSession({ diff --git a/packages/coding-agent/test/session-manager/file-operations.test.ts b/packages/coding-agent/test/session-manager/file-operations.test.ts index 9273a03a1..736ad72b7 100644 --- a/packages/coding-agent/test/session-manager/file-operations.test.ts +++ b/packages/coding-agent/test/session-manager/file-operations.test.ts @@ -3,9 +3,11 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { + type FileEntry, findMostRecentSession, loadEntriesFromFile, resolveResumableSession, + type SessionHeader, SessionManager, } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { getConfigRootDir, getSessionsDir, Snowflake, setAgentDir } from "@oh-my-pi/pi-utils"; @@ -292,3 +294,153 @@ describe("SessionManager temp cwd session dirs", () => { expect(fs.existsSync(path.join(expectedDir, "carried.jsonl"))).toBe(true); }); }); + +describe("SessionManager legacy session migration persistence", () => { + let tempDir: string; + + function makeAssistantMessage() { + return { + role: "assistant" as const, + content: [{ type: "text" as const, text: "legacy reply" }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-20250514", + usage: { + input: 1, + output: 1, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 2, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop" as const, + timestamp: Date.now(), + }; + } + + function getHeader(entries: FileEntry[]): SessionHeader | undefined { + return entries.find((entry): entry is SessionHeader => entry.type === "session"); + } + + beforeEach(() => { + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "omp-session-manager-legacy-")); + }); + + afterEach(() => { + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + it("keeps legacy migration in memory until later persisted activity rewrites the file", async () => { + const sessionFile = path.join(tempDir, "legacy.jsonl"); + fs.writeFileSync( + sessionFile, + `${[ + JSON.stringify({ type: "session", id: "legacy-session", timestamp: "2025-01-01T00:00:00Z", cwd: tempDir }), + JSON.stringify({ + type: "message", + timestamp: "2025-01-01T00:00:01Z", + message: { role: "user", content: "hello", timestamp: 1 }, + }), + JSON.stringify({ + type: "message", + timestamp: "2025-01-01T00:00:02Z", + message: makeAssistantMessage(), + }), + ].join("\n")}\n`, + ); + const initialMtimeMs = fs.statSync(sessionFile).mtimeMs; + + const session = await SessionManager.open(sessionFile, tempDir); + const migratedEntries = session.getEntries(); + + expect(migratedEntries).toHaveLength(2); + for (const entry of migratedEntries) { + expect(entry.id).toBeDefined(); + } + expect(migratedEntries[0]?.parentId).toBeNull(); + expect(migratedEntries[1]?.parentId).toBe(migratedEntries[0]?.id); + + await new Promise(resolve => setTimeout(resolve, 20)); + await session.flush(); + expect(fs.statSync(sessionFile).mtimeMs).toBe(initialMtimeMs); + + await new Promise(resolve => setTimeout(resolve, 20)); + session.appendMessage({ role: "user", content: "follow up", timestamp: Date.now() }); + await session.flush(); + + const persistedEntries = await loadEntriesFromFile(sessionFile); + const header = getHeader(persistedEntries); + if (!header) throw new Error("Expected session header"); + + expect(fs.statSync(sessionFile).mtimeMs).toBeGreaterThan(initialMtimeMs); + expect(header.version).toBe(3); + expect(persistedEntries).toHaveLength(4); + for (const entry of persistedEntries.filter(entry => entry.type !== "session")) { + expect(entry.id).toBeDefined(); + } + }); + + it("still rewrites immediately when explicitly requested", async () => { + const sessionFile = path.join(tempDir, "legacy-rewrite.jsonl"); + fs.writeFileSync( + sessionFile, + `${[ + JSON.stringify({ type: "session", id: "legacy-session", timestamp: "2025-01-01T00:00:00Z", cwd: tempDir }), + JSON.stringify({ + type: "message", + timestamp: "2025-01-01T00:00:01Z", + message: { role: "user", content: "hello", timestamp: 1 }, + }), + ].join("\n")}\n`, + ); + const initialMtimeMs = fs.statSync(sessionFile).mtimeMs; + + const session = await SessionManager.open(sessionFile, tempDir); + await new Promise(resolve => setTimeout(resolve, 20)); + await session.rewriteEntries(); + + const persistedEntries = await loadEntriesFromFile(sessionFile); + const header = getHeader(persistedEntries); + if (!header) throw new Error("Expected session header"); + + expect(fs.statSync(sessionFile).mtimeMs).toBeGreaterThan(initialMtimeMs); + expect(header.version).toBe(3); + expect(persistedEntries).toHaveLength(2); + expect(persistedEntries[1]?.type).toBe("message"); + if (persistedEntries[1]?.type !== "message") throw new Error("Expected message entry"); + expect(persistedEntries[1].id).toBeDefined(); + expect(persistedEntries[1].parentId).toBeNull(); + }); + + it("forces a deferred legacy rewrite when ensureOnDisk is requested", async () => { + const sessionFile = path.join(tempDir, "legacy-ensure-on-disk.jsonl"); + fs.writeFileSync( + sessionFile, + `${[ + JSON.stringify({ type: "session", id: "legacy-session", timestamp: "2025-01-01T00:00:00Z", cwd: tempDir }), + JSON.stringify({ + type: "message", + timestamp: "2025-01-01T00:00:01Z", + message: { role: "user", content: "hello", timestamp: 1 }, + }), + ].join("\n")}\n`, + ); + const initialMtimeMs = fs.statSync(sessionFile).mtimeMs; + + const session = await SessionManager.open(sessionFile, tempDir); + await new Promise(resolve => setTimeout(resolve, 20)); + await session.ensureOnDisk(); + + const persistedEntries = await loadEntriesFromFile(sessionFile); + const header = getHeader(persistedEntries); + if (!header) throw new Error("Expected session header"); + + expect(fs.statSync(sessionFile).mtimeMs).toBeGreaterThan(initialMtimeMs); + expect(header.version).toBe(3); + expect(persistedEntries).toHaveLength(2); + expect(persistedEntries[1]?.type).toBe("message"); + if (persistedEntries[1]?.type !== "message") throw new Error("Expected message entry"); + expect(persistedEntries[1].id).toBeDefined(); + expect(persistedEntries[1].parentId).toBeNull(); + }); +}); From 6d3a7dfbb2735be0e39dc5a55a1028e44a185357 Mon Sep 17 00:00:00 2001 From: luke <16418011+DeprecatedLuke@users.noreply.github.com> Date: Wed, 25 Mar 2026 13:06:01 +0000 Subject: [PATCH 22/22] fix: prevent subshell output read from blocking async runtime (#502) Move std::io::read_to_string into spawn_blocking and use tokio::join! to await both the output reader and the command concurrently. Prevents the async runtime from stalling when a child process holds the pipe open (e.g. a hung or slow external process). --- crates/brush-core-vendored/src/commands.rs | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/crates/brush-core-vendored/src/commands.rs b/crates/brush-core-vendored/src/commands.rs index 7b8b9c856..e7426f1ae 100644 --- a/crates/brush-core-vendored/src/commands.rs +++ b/crates/brush-core-vendored/src/commands.rs @@ -576,12 +576,18 @@ pub(crate) async fn invoke_command_in_subshell_and_get_output( rt.block_on(run_substitution_command(subshell, params, s)) }); - // Extract output. - let output_str = std::io::read_to_string(reader)?; + // Read subshell output on a blocking thread to avoid stalling the + // async runtime when the pipe stays open (e.g. a hung child process). + let output_join_handle = tokio::task::spawn_blocking(move || { + std::io::read_to_string(reader) + }); - // Now observe the command's completion. - let run_result = cmd_join_handle.await?; - let cmd_result = run_result?; + // Wait for both the output reader and the command to complete. + let (output_result, cmd_result) = tokio::join!(output_join_handle, cmd_join_handle); + let output_str = output_result + .map_err(|e| std::io::Error::other(e))??; + let cmd_result = cmd_result + .map_err(|e| std::io::Error::other(e))??; // Store the status. *shell.last_exit_status_mut() = cmd_result.exit_code.into();