diff --git a/packages/coding-agent/src/prompts/tools/ast-edit.md b/packages/coding-agent/src/prompts/tools/ast-edit.md
index 78bd577b6..7627819fd 100644
--- a/packages/coding-agent/src/prompts/tools/ast-edit.md
+++ b/packages/coding-agent/src/prompts/tools/ast-edit.md
@@ -1,22 +1,10 @@
-Structural AST-aware rewrites via ast-grep.
+Structural AST-aware rewrites via ast-grep. Use for codemods where text replace is unsafe. Narrow each call to one language.
-
-- Use for codemods / structural rewrites where text replace is unsafe
-- Narrow each call to one language
-- Metavariables captured in `pat` (`$A`, `$$$ARGS`) substitute into that entry's `out` template
-- **Patterns match AST structure, not text.** `$NAME` = one node (captured); `$_` = one without binding; `$$$NAME` = zero-or-more; `$$$` = zero-or-more without binding. Use `$$$NAME`, NOT `$$NAME` — the two-dollar form is invalid. Metavariable names are UPPERCASE and MUST be the whole AST node — partial text like `prefix$VAR` or `"hello $NAME"` does NOT work
-- Same metavariable twice → both occurrences MUST match identical code (`$A == $A` matches `x == x`, not `x == y`)
-- Rewrite patterns MUST parse as a single valid AST node. Non-standalone snippets → wrap in context, e.g. `class $_ { … }`
-- TS declarations/methods — tolerate unknown annotations: `async function $NAME($$$ARGS): $_ { $$$BODY }` or `class $_ { method($ARG: $_): $_ { $$$BODY } }`
-- Delete matched code with empty `out`: `{"pat":"console.log($$$)","out":""}`
-- Each rewrite is a 1:1 substitution — no splitting a capture across nodes or merging captures
-
-
-
-
-
-- Parse issues mean the rewrite is malformed or mis-scoped — fix the pattern before assuming a clean no-op
-- For one-off local text edits, you SHOULD prefer the Edit tool
-
+- Metavariables in `pat` (`$A`, `$$$ARGS`) substitute into `out`.
+- **Patterns match AST structure, not text.** `$NAME` = one node; `$_` = unbound; `$$$NAME` = zero-or-more.
+ - Use `$$$NAME`, NOT `$$NAME` (invalid). Names UPPERCASE, whole node — partial like `prefix$VAR` fails.
+- Same metavariable twice → MUST match identical code (`$A == $A` matches `x == x`, not `x == y`).
+- Rewrite patterns MUST parse as single AST node. Non-standalone → wrap: `class $_ { … }`.
+- TS: tolerate annotations — `async function $NAME($$$ARGS): $_ { $$$BODY }`. Delete with empty `out`: `{"pat":"console.log($$$)","out":""}`.
+- 1:1 substitution — no splitting/merging captures.
+- Parse issues → malformed rewrite, not clean no-op. For one-off text edits, prefer the Edit tool.
diff --git a/packages/coding-agent/src/prompts/tools/ast-grep.md b/packages/coding-agent/src/prompts/tools/ast-grep.md
index 8948637a5..e7ee59f86 100644
--- a/packages/coding-agent/src/prompts/tools/ast-grep.md
+++ b/packages/coding-agent/src/prompts/tools/ast-grep.md
@@ -1,25 +1,19 @@
-Structural code search via ast-grep.
+Structural code search via ast-grep. Use when syntax shape matters more than text (calls, declarations, language constructs).
-- Use when syntax shape matters more than text (calls, declarations, language constructs)
-- Narrow each call to one language
-- `pat` is ONE AST pattern; separate calls for unrelated patterns
-- `$NAME` captures one node; `$_` matches one without binding; `$$$NAME` captures zero-or-more; `$$$` matches zero-or-more without binding. Use `$$$NAME`, NOT `$$NAME` — the two-dollar form is invalid
-- Metavariable names are UPPERCASE and MUST be the whole AST node — partial text like `prefix$VAR`, `"hello $NAME"`, or `a $OP b` does NOT work
-- Same metavariable twice → both occurrences MUST match identical code (`$A == $A` matches `x == x`, not `x == y`)
-- Patterns MUST parse as a single valid AST node. Non-standalone snippets → wrap in context, e.g. `class $_ { … }`
-- C++ expression-statement calls need trailing `;`: `ns::doThing($ARG);`, `$CALLEE($ARG);`
-- TS declarations/methods — tolerate unknown annotations: `async function $NAME($$$ARGS): $_ { $$$BODY }` or `class $_ { method($ARG: $_): $_ { $$$BODY } }`
-- Declaration forms are distinct shapes — `function foo`, method `foo()`, `const foo = () => {}`; search the right form before concluding absence
-- Loosest existence check: `pat: "executeBash"` with narrow `path`
+- Narrow each call to one language. `pat` is ONE AST pattern; separate calls for unrelated patterns.
+- `$NAME` captures one node; `$_` matches without binding; `$$$NAME` zero-or-more; `$$$` zero-or-more unbound.
+ - Use `$$$NAME`, NOT `$$NAME` (invalid). Names UPPERCASE, whole node — `prefix$VAR` fails.
+- Same metavariable twice → MUST match identical code (`$A == $A` matches `x == x`, not `x == y`).
+- Patterns MUST parse as single AST node. Non-standalone → wrap: `class $_ { … }`.
+- C++ expression-statement calls need trailing `;`: `ns::doThing($ARG);`, `$CALLEE($ARG);`.
+- TS: tolerate annotations — `async function $NAME($$$ARGS): $_ { $$$BODY }`.
+- Declaration forms are distinct — `function foo`, method `foo()`, `const foo = () => {}`; search the right form before concluding absence.
+- Loosest existence check: `pat: "executeBash"` with narrow `path`.
-
-
-- AVOID repo-root scans — narrow `path` first
-- Parse issues = query failure, not absence: fix the pattern or tighten `path` before concluding "no matches"
-- Broad cross-subsystem exploration: you SHOULD use the Task tool + scout subagent first
+- AVOID repo-root scans — narrow `path` first.
+- Parse issues = query failure, not absence: fix pattern or tighten `path` before concluding "no matches".
+- Broad cross-subsystem exploration → Task tool + scout subagent first.
diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md
index fd5b6921b..931dfc587 100644
--- a/packages/coding-agent/src/prompts/tools/bash.md
+++ b/packages/coding-agent/src/prompts/tools/bash.md
@@ -1,72 +1,25 @@
-Runs commands in the embedded shell — terminal ops: git, bun, cargo, python.
+Runs commands in the embedded shell. NOT full GNU Bash — invokes real binaries with simple args.
-# When to use bash — and when not to
-
-The shell invokes **real binaries** with simple args. It is NOT full GNU Bash.
-
-Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact and does not depend on shell-specific regex/quoting (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`).
-{{#if hasLaunch}}Long-running service, watcher, debugger, REPL, or process needing later input? MUST use `launch`, not bash.{{/if}}
-
-{{#if hasEval}}Anything below → `eval` cell, not bash:
-- Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language
-- Heredocs (`<
-- `cwd` sets the working dir, not `cd dir && …`
-- `env: { NAME: "…" }` for multiline / quote-heavy / untrusted values; reference `$NAME`
-- Quote expansions (`"$NAME"`) to preserve exact content
-- `pty: true` only when the command needs a real terminal (`sudo`, `ssh` needing input); default `false`
-- `;` only when later commands should run despite earlier failures
-- Multiple bash calls per message run concurrently. NEVER split order-dependent commands across parallel calls — chain with `&&` in one call.
-- Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths
-{{#if hasEval}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.{{else}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Use a checked-in script, purpose-built tool, or single command that owns the output shape.{{/if}}
-{{#if asyncEnabled}}
-- `async: true` defers reporting for finite commands that need no later input; completion arrives as a follow-up.
-{{/if}}
+- `cwd` sets working dir (not `cd dir && …`). `env: { NAME: "…" }` for multiline/quote-heavy values; `"$NAME"` to expand.
+- `pty: true` only for real terminal needs (`sudo`, `ssh`); default `false`.
+- Multiple calls run concurrently; NEVER split order-dependent commands — chain with `&&` in one call (`;` only to continue past failure).
+- Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths.
+{{#if asyncEnabled}}- `async: true` defers reporting for finite commands needing no later input.{{/if}}
-{{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}}
-{{#if hasGrep}}- NEVER shell out to search content or files: `grep/rg` → `grep`.{{else}}- Avoid shelling out for broad content search; use an active search/read tool when one is available.{{/if}}
-{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.{{else}}- Prefer `read` for known file and directory reads. Only use shell listing when no file-listing tool is active.{{/if}}{{else}}{{#if hasGlob}}- Prefer `glob` for file discovery; avoid `find` when `glob` is active.{{else}}- If no file read/listing tool is active, keep shell inspection narrow and state that limitation.{{/if}}{{/if}}
-- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`.
-{{#if hasLaunch}}- NEVER launch daemons, watchers, dev servers, debuggers, or REPLs through bash/background shell syntax — use `launch`.{{/if}}
+{{#if hasGrep}}- NEVER shell out to search: `grep`/`rg` → built-in `grep`.{{/if}}
+{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` — `ls` → `read`, `find` → `glob`. NON-NEGOTIABLE.{{/if}}{{/if}}
+- Avoid head/tail/redirections: stderr merged, output auto-truncated, full capture at `artifact://`.
+{{#if hasLaunch}}- NEVER launch daemons/watchers/servers/debuggers/REPLs through bash — use `launch`.{{/if}}
-
-
-{{#if asyncEnabled}}
-# Timeout and async
-
-- `timeout` is seconds; nonzero values are clamped to `1..3600` and the process is killed on elapse. Set `timeout: 0` only for finite commands whose completion is cancellation-owned.
-- `async: true` defers only reporting; it does NOT extend a nonzero timeout.
-{{#if hasLaunch}}- Need a service, watcher, debugger, REPL, or later stdin? MUST use `launch`. NEVER use `cmd &`, `nohup`, or async bash as a process supervisor.{{else}}- Need a long-running process or >3600s run? Use an external process supervisor; avoid detached shell jobs you cannot later observe or stop.{{/if}}
-{{/if}}
-{{#if autoBackgroundEnabled}}
-
-## Auto-background
-
-- A long-running foreground call may convert to a background job; the final result arrives as a follow-up tool call. NOT a failure — don't retry or wait synchronously.
-- Need the result inline (e.g. piping into another command)? Raise `timeout` above expected duration{{#if asyncEnabled}}, or set `async: true` up front{{/if}}.
-{{/if}}
-
-# Output minimizer
-
-- Long output truncated; test/lint runner output filtered to failures. When visible text changed, a `[raw output: artifact://]` footer links the full capture — read it if a run looks suspicious or you need exact bytes.
-- No footer = what you see is exactly what the command emitted.
+{{#if asyncEnabled}}- `timeout`: nonzero clamped 1–3600, killed on elapse. `0` only for cancellation-owned. `async: true` defers reporting only, doesn't extend timeout.{{/if}}
+{{#if autoBackgroundEnabled}}- Long foreground calls may auto-background; result arrives as follow-up — NOT a failure. Need inline? Raise timeout{{#if asyncEnabled}} or `async: true`{{/if}}.{{/if}}
+- Long output truncated, test/lint filtered to failures. `[raw output: artifact://]` footer links full capture. No footer = what you see is exact output.
diff --git a/packages/coding-agent/src/prompts/tools/browser.md b/packages/coding-agent/src/prompts/tools/browser.md
index 27121f2b9..efc1b3f12 100644
--- a/packages/coding-agent/src/prompts/tools/browser.md
+++ b/packages/coding-agent/src/prompts/tools/browser.md
@@ -1,45 +1,26 @@
Drives real Chromium tab; full puppeteer access via JS.
-- Static content (articles, docs, issues/PRs, JSON, PDFs, feeds)? `read` the URL. Browser only for JS execution, auth, interactive actions.
-- Three actions:
- - `open` — acquire/reuse named tab (`name` defaults `"main"`). Optional `url` (navigate once ready), `viewport`, `dialogs: "accept" | "dismiss"` (auto-handle `alert`/`confirm`/`beforeunload`; else page hangs till you wire `page.on('dialog', …)`).
- - `close` — release tab by `name`, or all with `all: true`. `kill: true` also kills spawned-app process trees.
- - `run` — execute JS in existing tab. `code` = async function body; `page`, `browser`, `tab`, `display`, `assert`, `wait` in scope. Return value JSON-stringified into result; `display(value)` accumulates text/images. `wait(ms)` sleeps; `wait(fn, { timeout?, interval? })` polls `fn` (sync or async) until truthy and resolves with that value (default 100ms interval; deadline min(30s, cell budget − 1s), named error on timeout) — use it instead of in-page polling Promises inside `tab.evaluate`.
-- Tabs survive `run` calls and in-process subagents — open once, reuse.
-- Browser kinds (`app` on `open`):
- - default (no `app`) → headless Chromium with stealth patches.
- - `app.path` → spawn absolute binary (Electron/CDP). No stealth patches — NEVER tamper with a real desktop app.
- - `app.cdp_url` → connect to existing CDP endpoint (e.g. `http://127.0.0.1:9222`).
- - `app.target` (with `path`/`cdp_url`) — substring on url+title picks BrowserWindow.
-- `tab` helpers; drop to raw puppeteer `page` for anything uncovered:
- - `tab.goto(url, { waitUntil? })` — navigate. A hung load fails ~1s before the cell budget with a named, catchable error and the pending navigation is stopped; for slow pages raise `timeout` or use `waitUntil: "domcontentloaded"`.
- - `tab.observe({ includeAll?, viewportOnly? })` — accessibility snapshot: `{ url, title, viewport, scroll, elements: [{ id, role, name, value, states, … }] }`. Ids stable until next observe/goto.
- - `tab.ariaSnapshot(selector?, { depth?, boxes? })` — Playwright-format ARIA-tree YAML (nested roles + accessible names + `/url`/`/placeholder`), scoped to `selector` or the whole document. Every node carries a `[ref=eN]` id; `[cursor=pointer]` flags clickables. Captures dense, hierarchical structure/text that `observe()`'s flat list flattens away. Refs renumber from e1 each call and stay valid until the next `ariaSnapshot()`.
- - `tab.ref("e5")` — `[ref=eN]` from the last ariaSnapshot → element handle with the common action methods (`.click()`, `.type()`, `.fill()`, `.hover()`, `.evaluate()`, …); the primary way to act on a ref. For convenience `aria-ref=e5` also works inline in `tab.click`/`type`/`fill`/`waitFor`/`scrollIntoView` (e.g. `tab.click("aria-ref=e5")`).
- - `tab.id(n)` — id from last observe → element handle with the same action methods (`.click()`, `.type()`, `.fill()`, …).
- - `tab.click(selector)` / `tab.type(selector, text)` / `tab.fill(selector, value)` / `tab.press(key, { selector? })` / `tab.scroll(dx, dy)`.
- - `tab.waitFor(selector, { timeout? })` / `tab.waitForSelector(selector, { timeout?, visible?, hidden? })` — wait until attached (optionally visible/hidden); returns an action-method handle.
- - `tab.drag(from, to)` — endpoints: selector (center-to-center) or `{ x, y }` viewport point (canvases, sliders).
- - `tab.scrollIntoView(selector)` — center in viewport; before clicking off-screen elements.
- - `tab.select(selector, …values)` — set `
-- MUST `open` before `run` — `run` never creates a tab.
-- Default to `tab.observe()` for page state — structured data, actionable ids. Screenshot ONLY when appearance matters.
-- Navigation invalidates element ids — re-observe before use.
-- `code` runs with full Node access. Treat as your code, not sandboxed.
+- MUST `open` before `run`. Default to `tab.observe()`; screenshot only for appearance. `code` runs with full Node access — not sandboxed.
-
-
diff --git a/packages/coding-agent/src/prompts/tools/debug.md b/packages/coding-agent/src/prompts/tools/debug.md
index 3412066ae..b39b65940 100644
--- a/packages/coding-agent/src/prompts/tools/debug.md
+++ b/packages/coding-agent/src/prompts/tools/debug.md
@@ -1,17 +1,8 @@
-Debugger access.
+Debugger access. Prefer over bash for program state, breakpoints, stepping, or thread inspection.
-
-- You SHOULD prefer this over bash for program state, breakpoints, stepping, thread inspection, or interrupting a running process.
-- `action: "launch"` starts a session; `program` required, `adapter` optional. Python: `program` = target `.py`, interpreter/script flags in `args`. Go: `program` = package directory, `.go` file, or compiled binary.
-- `action: "attach"` connects to a running process: `pid` (local), `port` (remote), `adapter` forces a specific debugger.
-- **Breakpoints**: `set_breakpoint`/`remove_breakpoint` with source (`file`+`line`) or function (`function`); optional `condition`.
-- **Flow control**: `continue` (resume), `step_over`/`step_in`/`step_out` (single-step), `pause` (interrupt a running program).
-- **Inspect**: `threads`, `stack_trace` (current stopped thread), `scopes` (needs `frame_id` or current stopped frame), `variables` (needs `variable_ref` or `scope_id`), `evaluate` (needs `expression`; `context: "repl"` for raw debugger commands), `output` (stdout/stderr/console), `sessions`, `terminate`.
-
+Only one active session at a time. `program` is a target path, not a shell command. Directories need a directory-capable adapter (`dlv`).
-
-- Only one active debug session at a time.
-- `adapter` is a configured id: `gdb`, `lldb-dap`, `debugpy`, `dlv`, `rdbg`, or any `dap.json` entry; its command must be installed.
-- `program` is a target path, not a shell command. Directories require a directory-capable adapter such as `dlv`.
-- Python requires `debugpy` (`pip install debugpy`); Go requires Delve (`go install github.com/go-delve/delve/cmd/dlv@latest`); Ruby requires `rdbg` (`gem install debug`).
-
+Adapters:
+- Python: `debugpy` (`pip install debugpy`)
+- Go: Delve (`go install github.com/go-delve/delve/cmd/dlv@latest`)
+- Ruby: `rdbg` (`gem install debug`)
diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md
index 47c477983..8ba73935c 100644
--- a/packages/coding-agent/src/prompts/tools/eval.md
+++ b/packages/coding-agent/src/prompts/tools/eval.md
@@ -1,72 +1,42 @@
-Run one step of code in a persistent kernel.
+Run one step of code in a persistent kernel. State persists across calls and subagents.
-
-**One eval call = one cell = one logical step.** State persists per language across separate eval calls, tool calls, and `task` subagents — define helpers/datasets/clients in one call, then later calls reuse them directly.
+Work incrementally: imports → define → test → use, each its own cell. Re-run setup ONLY after `reset`, kernel crash.
+Parallelize *within* a cell with `parallel(thunks)`, not by batching.
-Work incrementally: imports in one call, define in the next, test, then use — each its own eval call. Re-run setup ONLY after `reset`, a kernel crash, or a `NameError`/`ReferenceError` proving the state is gone. Parallelize work *within* a cell with the `parallel(thunks)` helper, not by batching steps.
+{{#if py}}Top-level `await` works; `asyncio.run(…)` raises error.{{/if}}
+{{#if js}}JS runs under **Bun**: globals (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`) available; top-level `await`/`return` work.{{/if}}
-Fields:
-
-- `language` — {{#if py}}`"py"` IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` persistent JavaScript VM{{/if}}{{#if rb}}{{#ifAny py js}}, {{/ifAny}}`"rb"` persistent Ruby kernel{{/if}}{{#if jl}}{{#ifAny py js rb}}, {{/ifAny}}`"jl"` persistent Julia kernel{{/if}}.
-- `code` — cell body, verbatim. Newlines/quotes JSON-encoded; no fences, no headers.
-- `title` (optional) — short transcript label (e.g. `"imports"`).
-- `timeout` (optional) — seconds; `0` disables the cell timeout. Raise only for heavy compute or long non-agent tool calls.
-- `reset` (optional) — wipe this language's kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
-
-{{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}}
-{{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}}
-{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `output("id", limit: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}}
-{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `output("id", limit=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}}
-On error, fix and re-run only the failing step — prior calls' state survives.
-
+On error, fix and re-run only the failing step.
-{{#ifAll py js}}Same helpers + arg order, both runtimes. Python: sync, options = trailing kwargs. JS: async/`await`able, options = ONE trailing object literal, never positional (extras throw).{{else}}{{#if py}}Sync; options = trailing kwargs.{{/if}}{{#if js}}Async/`await`able; options = ONE trailing object literal, never positional (extras throw).{{/if}}{{/ifAll}}{{#if rb}} Ruby: sync, options = trailing keyword args.{{/if}}{{#if jl}} Julia: sync, options = trailing keyword args.{{/if}}
+{{#ifAll py js}}Python: sync, kwargs. JS: async, ONE trailing object literal, never positional.{{else}}{{#if py}}Sync; kwargs.{{/if}}{{#if js}}Async; ONE trailing object literal, never positional.{{/if}}{{/ifAll}}{{#if rb}} Ruby: sync, kwargs.{{/if}}{{#if jl}} Julia: sync, kwargs.{{/if}}
```
-display(value) → None
- Cell output; figures/images/dataframes shown natively.
-print(value, ...) → None
- Text output.
+display(value) → None print(value, ...) → None
read(path, offset?=1, limit?=None) → str
- File/resource text; offset/limit = 1-indexed lines. `local://…` works everywhere; Python/JS also accept top-level `read` URI schemes.
write(path, content) → str
- Write file (creates parents) → resolved path. `local://…` persists across turns/subagents.
env(key?=None, value?=None) → str | None | dict
- No args → full env dict; one → value of `key`; two → set `key=value`, return value.
output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | dict | list[dict]
- Task/agent output by id; one → text/dict, multiple → list.
tool.(args) → unknown
- Invoke any session tool; `args` = its parameter object.
-completion(prompt, model?="default", system?=None, schema?=None) → str | dict
- Oneshot, stateless (no history/tools). `model`: "smol" fast | "default" session | "slow" most capable. `schema` (JSON-Schema) → structured output, parsed object.
-{{#if spawns}}agent(prompt, agent?="{{spawnDefaultAgent}}", model?=None, label?=None, schema?=None, handle?=False) → str | dict
- Run a subagent → final output. `agent` picks another discovered agent; omit it to use `{{spawnDefaultAgent}}`.{{#if spawnAllowedAgentsText}} Allowed agents: {{spawnAllowedAgentsText}}.{{/if}} `schema` as in completion(). Background via `local://` files named in the prompt. `handle` → DAG node dict { text, output, handle: "agent://", id, agent } (parsed under `data` when `schema` set).
-{{#if js}} JS: options are ONE trailing object — agent(prompt, { agent, schema, handle }).
+completion(prompt, model?="default"|"smol"|"slow", system?=None, schema?=None) → str | dict
+{{#if spawns}}agent(prompt, agent?="{{spawnDefaultAgent}}", model?=None, schema?=None, handle?=False) → str | dict{{#if spawnAllowedAgentsText}} Allowed: {{spawnAllowedAgentsText}}.{{/if}}
+{{#if js}} JS: agent(prompt, { agent, schema, handle }).{{/if}}
{{/if}}
-{{/if}}
-parallel(thunks) → list
- Thunks through a bounded pool (wide as a `task` batch — don't pre-shrink), input order kept; returns when all finish, a throwing thunk propagates.
-pipeline(items, ...stages) → list
- Map items through one-arg stages left-to-right, barrier between stages; stage 1 gets the item, later stages the previous result.
-log(message) → None
- Progress line above the status tree.
-phase(title) → None
- Phase grouping subsequent status lines.
-budget → per-turn token budget
- {{#if py}}`budget.total` (ceiling or None), `budget.spent()`, `budget.remaining()` (math.inf when no ceiling), `budget.hard`.{{/if}}{{#if js}}`await budget.total()` (ceiling or null), `await budget.spent()`, `await budget.remaining()` (Infinity when no ceiling), `await budget.hard()`.{{/if}}{{#if rb}} Ruby: `budget.total` (ceiling or nil), `budget.spent`, `budget.remaining` (Float::INFINITY when no ceiling), `budget.hard`.{{/if}}{{#if jl}} Julia: `budget.total` (ceiling or nothing), `budget.spent()`, `budget.remaining()` (Inf when no ceiling), `budget.hard`.{{/if}} Ceiling: `+Nk` (advisory) or `+Nk!`/Goal Mode (hard — `agent()` won't spawn past it); spend still tracked.
+parallel(thunks) → list pipeline(items, ...stages) → list
+log(message) → None phase(title) → None
+budget → {{#if py}}`budget.total` (ceiling or None), `budget.spent()`, `budget.remaining()`{{/if}}{{#if js}}`await budget.total()`, `await budget.spent()`, `await budget.remaining()`{{/if}}{{#if rb}}`budget.total`, `budget.spent`, `budget.remaining`{{/if}}{{#if jl}}`budget.total`, `budget.spent()`, `budget.remaining()`{{/if}}; ceiling `+Nk` advisory, `+Nk!` hard.
```
{{#if spawns}}
-Pipe handles through stage helpers to build a dependency graph — acyclic waves:
-- **Name nodes.** Capture each `agent(…, {{#if py}}handle=True{{/if}}{{#if js}}{ handle: true }{{/if}}{{#if jl}}handle=true{{/if}})` result; carries `handle` (`agent://`) + `output`.
-- **Wire edges by reference.** Put an upstream node's `handle`/`output` in the dependent stage's prompt — large transcript never re-inlined. Bulk: `write("local://.md", …)`, pass the URI.
-- **`pipeline(items, *stages)` = staged waves**, barrier between stages (every item clears stage N before any enters N+1). **`parallel(thunks)` = one wave** of independent nodes.
-- **Isolate failure.** A raising node re-raises the lowest-index error, aborts its wave; wrap risky nodes in try/except so a failure degrades only its dependent subtree, independent branches finish.
-- **Acyclic only.** A node never waits on its own descendant.
+Acyclic waves via `agent(…, handle=true)` + `pipeline`/`parallel`:
+- **Name nodes.** Capture agent result → `handle` (`agent://`) + `output`.
+- **Wire edges.** Put upstream `handle`/`output` in downstream prompt. Bulk: `write("local://.md", …)`.
+- **`pipeline`** = staged waves, barrier between stages. **`parallel`** = one wave.
+- **Isolate failure.** Wrap risky nodes in try/except; a failure degrades only its subtree.
+- **Acyclic only.** No node waits on its own descendant.
{{/if}}
-Prior top-level names (`data`, `sessions`, helpers, imports) survive into the next eval call — reuse them; NEVER re-import, re-require, or re-declare a helper. Re-read a file only if it may have changed since the last read. Re-run setup only after `reset`, a crash, or a `NameError`/`ReferenceError`.
+Prior top-level names survive into the next cell — reuse; NEVER re-import/re-declare. Re-read only if file changed since last read.
diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md
index 76c9b52b2..04f7f4c8e 100644
--- a/packages/coding-agent/src/prompts/tools/grep.md
+++ b/packages/coding-agent/src/prompts/tools/grep.md
@@ -1,22 +1,12 @@
-Greps files using regex.
+Greps files using regex (Rust regex + PCRE2).
-- Supports Rust regex and PCRE2 syntax.
-- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Append a line selector to one file path (e.g. `src/foo.ts:50-100`); selectors never choose the search root.
-- Cross-line patterns detected from literal `\n` or `\\n` in `pattern`.
+- `path`: scope to known path (e.g. `src`); pass several as delimited list (`src; tests`).
+ Line selector on one file (`src/foo.ts:50-100`); selectors never choose search root.
+- Cross-line patterns from literal `\n` or `\\n` in `pattern`.
-
-
-- MUST use built-in `grep` for any content search. NEVER shell out to `grep`, `rg`, `ripgrep`, `ag`, `ack`, `git grep`, `awk`, `sed`-for-search, or any CLI search via Bash — not even for one match or a quick check.
-- Open-ended search needing multiple rounds? MUST use the Task tool with the scout subagent, NOT chained `grep` calls.
+- MUST use this over bash when searching!
+- Open-ended multi-round search → Task tool + scout subagent, NOT chained `grep` calls.
diff --git a/packages/coding-agent/src/prompts/tools/image-gen.md b/packages/coding-agent/src/prompts/tools/image-gen.md
index 8e1d72f42..6058790c0 100644
--- a/packages/coding-agent/src/prompts/tools/image-gen.md
+++ b/packages/coding-agent/src/prompts/tools/image-gen.md
@@ -1,7 +1,7 @@
Generates or edits images.
-- You MUST provide a single detailed `subject` prompt for image generation or editing.
-- When using multiple `input`, you SHOULD describe each image's role directly in `subject`, e.g. `Image 1` for composition reference, `Image 2` for lighting reference, `Image 3` for background.
-- For text: you SHOULD add "sharp, legible, correctly spelled" for important text; keep text short.
+- Provide a single detailed `subject` prompt for generation or editing.
+- When using multiple `input`, describe each image's role in `subject` (e.g. `Image 1` for composition, `Image 2` for lighting).
+- For text: add "sharp, legible, correctly spelled"; keep text short.
diff --git a/packages/coding-agent/src/prompts/tools/irc.md b/packages/coding-agent/src/prompts/tools/irc.md
index 09e8619aa..1977670e5 100644
--- a/packages/coding-agent/src/prompts/tools/irc.md
+++ b/packages/coding-agent/src/prompts/tools/irc.md
@@ -1,24 +1,11 @@
-Send and receive short text messages between the agents running in this process.
+Agent-to-agent messaging. Main agent is `Main`; subagents inherit task ID.
+Use `op: "list"` to discover peers. Address by exact roster ID — NEVER invent names.
-# Addressing and Discovery
-The main agent is always `Main`. Subagents inherit their task ID (e.g., `AuthLoader`). If you don't know who is currently running, use `op: "list"` to view all peers alongside their status, unread message count, and recent activity. Address peers by their exact ID from the roster; NEVER invent names.
-
-# Messaging Rules
-Use `op: "send"` to deliver a message to a specific peer or broadcast to `"all"`.
-- **Fire and forget:** Sending NEVER blocks. You get delivery receipts immediately (`delivered` or `failed`). Do not wait around—send your message and keep working. If a receipt says `failed`, the peer is gone; do not retry.
-- **Waking peers:** Sending a message to an `idle` or `parked` agent automatically wakes them up.
-- **Answering:** When replying to a question, use `op: "send"`, lead directly with your answer (NEVER quote the original message), and set `replyTo` so the recipient can correlate it.
-- **Format:** Messages MUST be plain prose. NEVER send JSON status objects. Keep it terse and share paths via `local://` or `artifact://` URLs, not pasted blobs.
-
-# Waiting and Inboxes
-Messages only arrive when the peer actively sends one—do not interrogate a peer for status.
-- If you are completely blocked and MUST wait for an answer, use `op: "wait"` (or `await: true` on a send). The wait returns when a matching message arrives, the timeout elapses, or any IRC / steering message interrupts the wait. Parent-agent IRC interrupts with steering-level priority.
-- No need to alternate `irc wait`, `irc inbox`, and `job poll`: waits surface cross-channel interrupts promptly. The next turn includes the interrupt reason and message.
-- To check for messages without blocking, use `op: "inbox"` to drain your queue.
-
-# When to Coordinate
-Message peers instead of guessing, duplicating work, or spying.
-- Use IRC when you hit an unexpected state (e.g., missing files) or an out-of-scope decision. DM `Main` or your spawner for guidance.
-- If you overlap with another agent's work or need a file they are touching, DM them before editing.
+- **`send`**: fire-and-forget, NEVER blocks. Delivery receipts (`delivered`/`failed`) immediate; `failed` → peer gone, don't retry.
+ Sending wakes `idle`/`parked` peers. Answering: lead with answer, NEVER quote, set `replyTo`.
+- **Format**: plain prose ONLY. No JSON status objects. Share paths via `local://`/`artifact://` URLs, not pasted blobs.
+- **`wait`** (or `await: true`): blocks until matching message, timeout, or steering interrupt. Parent IRC interrupts at steering priority.
+ Waits surface cross-channel interrupts — don't alternate `wait`/`inbox`/`job poll`.
+- **`inbox`**: drain queue without blocking.
- NEVER use shell tools, grep, or read other sessions' files to figure out what a peer is doing. Message them directly.
- NEVER use IRC for something a tool can answer (e.g., grepping codebase, running a build).
diff --git a/packages/coding-agent/src/prompts/tools/lsp.md b/packages/coding-agent/src/prompts/tools/lsp.md
index 341f32acf..83e0051e2 100644
--- a/packages/coding-agent/src/prompts/tools/lsp.md
+++ b/packages/coding-agent/src/prompts/tools/lsp.md
@@ -1,28 +1,19 @@
-Symbol-aware code intelligence from language servers — the accurate path for navigation, refactors, and diagnostics where text search or edits would miss callsites.
+Symbol-aware code intelligence from language servers — navigation, refactors, and diagnostics where text tools miss callsites.
-Position-based — pass `file` + `line` + `symbol` (substring on that line; append `#N` for the Nth match, e.g. `kind#2`):
-- `definition`, `type_definition`, `implementation`, `references`, `hover` — standard LSP lookups
-- `rename` — rename the symbol everywhere; **applies by default**, `apply: false` previews; needs `new_name`
-- `code_actions` — quick-fixes/refactors/imports at that position; lists by default (`query` filters by kind, e.g. `quickfix`, `source.organizeImports`), **applies one only with `apply: true` + `query`** (then `query` = action title substring or numeric index)
-
-File / workspace:
-- `diagnostics` — errors/warnings for a path, a glob (`src/**/*.ts`), or the whole workspace (`file: "*"`)
-- `symbols` — `file` lists that file's symbols; `file: "*"` + `query` searches the workspace
-- `rename_file` — move `file` → `new_name` on disk AND rewrite imports/references through the server; applies by default
-
-Servers:
-- `status`, `capabilities` — what's running / per-server capabilities (one via `file`, all via `*`)
-- `reload` — restart one server (`file`) or all (`*`); `reload *` also re-reads project LSP config
-- `request` — raw escape hatch: `query` = method (`rust-analyzer/expandMacro`, `workspace/executeCommand`), `payload` = JSON params (else auto-built from `file`/`line`/`symbol`)
+- Position-based: `file` + `line` + `symbol` (substring; `#N` for Nth match). `line` is 1-indexed.
+- `rename` — applies by default; `apply: false` previews. Project-aware lookups ERROR without `symbol` — no silent fallback on missing/ambiguous matches.
+- `code_actions` — lists by default; apply ONE with `apply: true` + `query` (title substring or index).
+- `rename_file` — moves file AND rewrites all imports/references; applies by default.
+- `diagnostics` — path, glob (`src/**/*.ts`), or `file: "*"` for workspace.
+- `symbols` — `file` lists file symbols; `file: "*"` + `query` searches workspace.
+- `reload` — restart one server (`file`) or all (`*`); `reload *` re-reads LSP config.
+- `request` — raw: `query` = method, `payload` = JSON params (else auto-built).
-
-- `line` is 1-indexed. Project-aware `definition`/`references`/`rename` ERROR without `symbol` rather than guess the wrong identifier; a missing match or out-of-range `#N` is an explicit error, never a silent fallback.
-
-
-- Symbol-aware work (rename, references, definition/type/impl, code actions) MUST use `lsp` whenever a server is available — it follows shadowing, re-exports, and cross-file usages that text tools miss.
-- NEVER do a cross-file rename with `ast_edit`, `sed`, or hand edits when `lsp` `rename`/`rename_file` can — text renames silently drop callsites.
+- Symbol-aware work (rename, references, definition, code actions) MUST use `lsp` whenever a server is available.
+ It follows shadowing, re-exports, and cross-file usages text tools miss.
+- NEVER do a cross-file rename with `ast_edit`/`sed`/hand edits when `lsp` `rename`/`rename_file` can — text renames silently drop callsites.
- Reach for `code_actions` on imports, quick-fixes, and server-known refactors before editing by hand.
diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md
index a4ccd0bc2..b2bcd639c 100644
--- a/packages/coding-agent/src/prompts/tools/read.md
+++ b/packages/coding-agent/src/prompts/tools/read.md
@@ -2,75 +2,26 @@ Read files, directories, archives, SQLite, images, documents, internal resources
- SHOULD parallelize independent reads.
-- SHOULD use `read` (not a browser tool) for web content; browser only when `read` can't deliver.
+- SHOULD use `read` (not browser) for web content; browser only when `read` can't deliver.
-## Parameters
-
-- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Append `:` for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`).
-
-## Selectors
-
-- _(none)_ — parseable code → structural summary; other files → from start (up to {{DEFAULT_LIMIT}} lines).
-- `:50` / `:50-` — from line 50 onward.
-- `:50-200` — lines 50–200 inclusive.
-- `:50+150` — 150 lines from 50.
-- `:20+1` — anchor line 20.
-- `:5-16,960-973` — multiple ranges in one call.
-- `:raw` — verbatim; no anchors/summary/line prefixes.
-- `:2-4:raw` / `:raw:2-4` — range AND verbatim; either order.
-- `:conflicts` — one line per unresolved git merge conflict block.
-
-# Files
+## Selectors — append `:` to `path` (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`)
+- `:50` / `:50-` — from line 50 | `:50-200` — inclusive | `:50+150` — 150 lines from 50 | `:5-16,960-973` — multiple ranges
+- `:raw` — verbatim, no anchors/prefixes | `:2-4:raw` / `:raw:2-4` — range + verbatim
+- `:conflicts` — one line per unresolved git merge conflict block
+## Source kinds
+- Parseable code, no selector → structural summary (declarations only, body elided). Footer names recovery selector — re-issue ONLY those ranges.
+- {{#if IS_HL_MODE}}File + selector → `[foo.ts#1A2B]` snapshot header + numbered lines. Copy `[FILENAME#TAG]` for anchored edits; NEVER fabricate the tag.{{/if}}
- Directory → depth-limited dirent listing.
-{{#if IS_HL_MODE}}
-- File + selector → filename-only snapshot header + numbered lines: `[foo.ts#1A2B]` then `41:def alpha():`. Copy `[FILENAME#TAG]` for anchored edits; ops use bare line numbers. NEVER fabricate the tag.
-{{else}}
-{{#if IS_LINE_NUMBER_MODE}}
-- File + selector → numbered lines: `41|def alpha():`.
-{{/if}}
-{{/if}}
-- Parseable code, no selector → **structural summary**: declarations kept, body elided with `…`. Footer names the recovery selector; re-issue ONLY the ranges you need.
-
-# Documents & Notebooks
-
-PDF, Word, PowerPoint, Excel, RTF, EPUB → extracted text. Notebooks (`.ipynb`) → editable `# %% [type] cell:N` text. `:raw` bypasses the converter.
-
-# Images
-
-{{#if INSPECT_IMAGE_ENABLED}}
-Image → metadata. Visual analysis: call `inspect_image` with the path and a question.
-{{else}}
-Image → decoded inline (PNG, JPEG, GIF, WEBP) for direct visual analysis.
-{{/if}}
-
-# Archives
-
-`.tar`, `.tar.gz`, `.tgz`, `.zip`. `archive.ext:path/inside/archive` reads a member; inner paths take normal selectors: `archive.zip:dir/file.ts:50-60`.
-
-# SQLite
-
-For `.sqlite`, `.sqlite3`, `.db`, `.db3`:
-- `file.db` — tables with row counts
-- `file.db:table` — schema + sample rows
-- `file.db:table:key` — row by primary key
-- `file.db:table?limit=50&offset=100` — pagination
-- `file.db:table?where=status='active'&order=created:desc` — filter/order
-- `file.db?q=SELECT …` — read-only SELECT
-
-# URLs
-
-- Reader-mode default: HTML, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom, JSON endpoints, PDFs → clean text/markdown.
-- `:raw` → untouched HTML; line selectors (`:50`, `:50-100`, `:50+150`) paginate the fetch.
-- Bare `host:port` collides with selector grammar — add a trailing slash: `https://example.com/:80`.
-
-# Internal URIs
-
-All URI schemes take the same line selectors. `artifact://` recovers spilled output; large artifacts block unbounded `:raw`, so page with `artifact://:N-M` / `artifact://:raw:N-M` and use the reported artifact file path for search/copy workflows.
-
-`ssh://host/` reads a remote text file (UTF-8, ≤1 MiB) or lists a directory one level deep, on a pre-configured SSH host or `~/.ssh/config` alias; `ssh://host/` lists the remote root and bare `ssh://` lists the configured hosts. Files are also writable via `write` and searchable via `search`; a directory only lists (`search` refuses a directory, `write` refuses to overwrite one). A literal `:`, `?`, or `#` in the remote path must be percent-encoded (`%3A`/`%3F`/`%23`) — a trailing `:sel` is read as a line selector, and `?`/`#` start a URL query/fragment. Requires a POSIX login shell (`sh`/`bash`/`zsh`); a Windows host or a non-POSIX shell (fish, csh/tcsh) is rejected — use the `ssh` tool there.
+- SQLite (`.sqlite`, `.sqlite3`, `.db`, `.db3`): `file.db` (tables), `file.db:table` (schema+rows), `file.db:table:key` (by PK), `?limit=`/`?where=`/`?q=SELECT`.
+- Archives (`.tar`, `.tar.gz`, `.tgz`, `.zip`): `archive.ext:path/inside/archive` reads a member.
+- Documents → extracted text. Notebooks → editable cells. Images → {{#if INSPECT_IMAGE_ENABLED}}metadata; call `inspect_image`{{else}}decoded inline{{/if}}. `:raw` bypasses converters.
+- URLs → reader-mode clean text/markdown; `:raw` → untouched HTML. Bare `host:port` needs trailing slash.
+- Internal URIs — all schemes take selectors. `artifact://` recovers spilled output; page with `:N-M`/`:raw:N-M`.
+- `ssh://host/` reads remote file/dir (UTF-8, ≤1 MiB); bare `ssh://` lists hosts; also `write`/`search`-able.
+ Literal `:`, `?`, `#` → percent-encode (`%3A`/`%3F`/`%23`). Requires POSIX shell (else `ssh` tool).
-- Summary footer names elided ranges? Re-issue ONLY those ranges. NEVER guess `..`/`…` content.
+Summary footer names elided ranges? Re-issue ONLY those ranges. NEVER guess `..`/`…` content.
diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md
index acf9ce4c1..d41cf71e3 100644
--- a/packages/coding-agent/src/prompts/tools/task.md
+++ b/packages/coding-agent/src/prompts/tools/task.md
@@ -1,50 +1,45 @@
-{{#if asyncEnabled}}{{#if batchEnabled}}Delegate work to background subagents by passing multiple items in a single `tasks[]` batch.{{else}}Delegate work to ONE background subagent per call.{{/if}}
-Execution does not block your turn: you receive agent and job IDs immediately, and the final results deliver themselves when the subagents finish.{{#if hasBlockingAgents}}
-Exception: agents marked BLOCKING below run inline — their results return in this call, while non-blocking items in the same batch still spawn as background jobs.{{/if}}{{else}}{{#if batchEnabled}}Run subagents synchronously by passing items in a `tasks[]` batch.{{else}}Run ONE subagent synchronously per call.{{/if}}
-Execution blocks your turn: the call only returns once the work is completely finished.{{/if}}
+{{#if asyncEnabled}}{{#if batchEnabled}}Delegate work to background subagents by passing multiple items in a single `tasks[]` batch.
+Execution does not block — you receive IDs immediately; results deliver when subagents finish.{{else}}Delegate work to ONE background subagent per call.
+Execution does not block — you receive an ID immediately; the result delivers when the subagent finishes.{{/if}}{{#if hasBlockingAgents}}
+Agents marked BLOCKING run inline — results return in this call; non-blocking items in the same batch still spawn as background jobs.{{/if}}{{else}}{{#if batchEnabled}}Run subagents synchronously by passing items in a `tasks[]` batch. Execution blocks until all work finishes.{{else}}Run ONE subagent synchronously. Execution blocks until work finishes.{{/if}}{{/if}}
# Task Design
-- **Agent typing:** Choose each item's `agent` type first. Read-only research MUST use `agent: "scout"`, which runs on a faster model. Use the default worker only when no listed specialist fits.
-- **No overhead:** Each `task` MUST instruct its agent to skip formatters, linters, and project-wide test suites. You will run those once at the end.
-- **One-pass agents:** Prefer agents that investigate **and** edit in a single pass; only spin a read-only discovery step (e.g. `agent: "scout"`) when the affected files are genuinely unknown.
+- **Agent typing:** Pick each item's `agent` type. Read-only research MUST use `agent: "scout"` (faster model). Use default worker only when no specialist fits.
+- **No overhead:** Each `task` MUST instruct its agent to skip formatters, linters, and project-wide test suites. Run those once at the end.
+- **One-pass:** Prefer agents that investigate AND edit in one pass; spin a read-only scout only when affected files are genuinely unknown.
# Inputs
{{#if batchEnabled}}
-- `context`: Shared project state, constraints, and contracts. Applies to the entire batch; do not duplicate this background into individual tasks.
-- `tasks[]`: Array of subagents to spawn.
- - `name`: A stable CamelCase identifier (≤32 chars), used to address the agent (IRC, job ids). Generated automatically if omitted.
- - `agent`: The agent type running this item (e.g. `scout`, `reviewer`). Omitting it gives you the general-purpose worker (`{{defaultAgent}}`) — NEVER pass that name explicitly. Only omit it after checking the agent list below and finding no specialist that fits.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}}
- - `task`: Complete, self-contained instructions. One-liners or missing acceptance criteria are PROHIBITED.
+- `context`: Shared project state for the entire batch — don't duplicate into individual tasks.
+- `tasks[]`: Subagents to spawn.
+ - `name`: CamelCase ≤32 chars (auto-generated if omitted).
+ - `agent`: specialist type (optional).
+ - `task`: Complete, self-contained instructions — no one-liners, no missing acceptance criteria.
{{#if isolationEnabled}}
- - `isolated`: Run in a dedicated worktree and return patches. Isolated agents are destroyed upon completion and cannot be addressed afterward.
+ - `isolated`: Run in dedicated worktree, return patches. Destroyed on completion, cannot be addressed afterward.
{{/if}}
{{else}}
-- `name`: A stable CamelCase identifier (≤32 chars), used to address the agent (IRC, job ids). Generated automatically if omitted.
-- `agent`: The agent type to spawn (e.g. `scout`, `reviewer`). Omitting it gives you the general-purpose worker (`{{defaultAgent}}`) — NEVER pass that name explicitly. Only omit it after checking the agent list below and finding no specialist that fits.{{#if allowedAgentsText}} Current spawn policy allows: {{allowedAgentsText}}.{{/if}}
-- `task`: Complete, self-contained instructions. One-liners or missing acceptance criteria are PROHIBITED.
+- `name`: CamelCase ≤32 chars (auto-generated if omitted).
+- `agent`: specialist type (optional).
+- `task`: Complete, self-contained instructions — no one-liners, no missing acceptance criteria.
{{#if isolationEnabled}}
-- `isolated`: Run in a dedicated worktree and return patches. Isolated agents are destroyed upon completion and cannot be addressed afterward.
+- `isolated`: Run in dedicated worktree, return patches.
{{/if}}
{{/if}}
-# Context and Communication
-Subagents start blank. They have no access to your conversation history.
-{{#if ircEnabled}}- **Steering delivery:** Parent-to-subagent IRC is delivered immediately as steering; subagents blocked in `job poll` / `irc wait` do not need to poll separately for it.{{/if}}
-{{#if batchEnabled}}
-- Pass large payloads using `local://` URIs, NEVER inline text.
-{{else}}
-- Write shared project state ONCE to a `local://` file (e.g., `local://ctx.md`) and reference that URL in each `task`.
-{{/if}}
+# Communication
+Subagents start blank — no conversation history.{{#if ircEnabled}} Parent-to-subagent IRC delivered immediately as steering.{{/if}}
+Pass large payloads via `local://` URIs, NEVER inline text.
# Format Contracts
{{#if batchEnabled}}
-The `context` field MUST follow this format:
+`context` format:
# Goal ← what the batch accomplishes
# Constraints ← rules and session decisions
# Contract ← shared interfaces
{{/if}}
-The `task` field MUST follow this format:
+`task` format:
# Target ← exact files and symbols; explicit non-goals
# Change ← step-by-step add/remove/rename; APIs and patterns
# Acceptance ← observable result; no project-wide commands
@@ -53,10 +48,10 @@ The `task` field MUST follow this format:
{{#if spawningDisabled}}
Agent spawning is currently disabled.
{{else}}
-Pick the most specific agent for each task. Use the default worker only when no specialist below fits.
+Pick the most specific agent; use default worker only when no specialist fits.
{{#list agents join="\n"}}
-### {{name}}{{#if readOnly}} (READ-ONLY: no edit/write/command tools){{/if}}{{#if blocking}} (BLOCKING: runs inline; its result returns in this call){{/if}}
+### {{name}}{{#if readOnly}} (READ-ONLY){{/if}}{{#if blocking}} (BLOCKING: inline result){{/if}}
{{description}}
-{{#if readOnly}}Use ONLY for investigation and reporting; do the edits yourself or assign them to a writing agent.{{/if}}
+{{#if readOnly}}Use ONLY for investigation; do edits yourself or assign to a writing agent.{{/if}}
{{/list}}
{{/if}}
diff --git a/packages/coding-agent/src/prompts/tools/todo.md b/packages/coding-agent/src/prompts/tools/todo.md
index 7be97f33d..2c5a61aeb 100644
--- a/packages/coding-agent/src/prompts/tools/todo.md
+++ b/packages/coding-agent/src/prompts/tools/todo.md
@@ -1,6 +1,7 @@
**Tasks referenced by verbatim content string, NEVER an auto-generated ID — no "task-1"/"task-N" exists. Pass the content text in the `task` field.**
-On each completion the earliest still-open task (in phase order) auto-promotes to `in_progress`. Completing tasks out of phase order can move this pointer **back** to an earlier phase — that is expected; completed tasks are never reverted.
+On each completion the earliest still-open task (in phase order) auto-promotes to `in_progress`.
+Completing tasks out of phase order can move this pointer **back** to an earlier phase — expected; completed tasks are never reverted.
## Operations
@@ -11,20 +12,19 @@ On each completion the earliest still-open task (in phase order) auto-promotes t
|`start`|`task`|Mark in progress|
|`done`|`task` or `phase`|Mark completed|
|`drop`|`task` or `phase`|Mark abandoned|
-|`rm`|`task` or `phase` (optional)|Remove task or phase's tasks; omit both to clear the list|
+|`rm`|`task` or `phase` (optional)|Remove task or phase; omit both to clear|
|`append`|`phase`, `items: string[]`|Append tasks to `phase`; lazily creates phase|
-|`view`|—|Read-only: echo the list, no modify|
+|`view`|—|Read-only: echo list|
## Anatomy
- **Task content**: 5–10 words; what, not how. Unique identifier.
- **Phase name**: short noun phrase (e.g. `Foundation`, `Auth`, `Verification`). Unique identifier. NEVER prefix `1.`, `A)`, `Phase 1:`.
## Rules
-- Mark tasks done immediately after finishing.
-- Complete phases in order.
-- Blocked? `append` a task to the active phase to unblock, or `drop`.
+- Mark tasks done immediately after finishing. Complete phases in order.
+- Blocked? `append` a task to the active phase, or `drop`.
- Keep `task`/`phase` strings stable once introduced.
-- Lost the exact task text? `view` echoes the list — NEVER guess from memory; a mismatched `task` string is an error.
+- Lost the exact task text? `view` echoes the list — NEVER guess from memory.
## When to create a list
- Task requires 3+ distinct steps
diff --git a/packages/coding-agent/test/tools/eval-description.test.ts b/packages/coding-agent/test/tools/eval-description.test.ts
index eb4288d48..d8351799a 100644
--- a/packages/coding-agent/test/tools/eval-description.test.ts
+++ b/packages/coding-agent/test/tools/eval-description.test.ts
@@ -21,13 +21,11 @@ function makeSession(opts: { spawns?: string | null; backends?: Record {
expect(wildcard).toContain("agent(prompt");
expect(denied).not.toContain("agent(prompt");
});
-
- it("documents zero as the unlimited timeout value", () => {
- const tool = new EvalTool(makeSession({}));
- const fields = wireCellFields(tool);
- expect(fields.timeoutDescription).toContain("0 disables the cell timeout");
- expect(tool.description).toContain("`0` disables the cell timeout");
- });
});
describe("eval tool dynamic schema", () => {
diff --git a/packages/coding-agent/test/tools/index.test.ts b/packages/coding-agent/test/tools/index.test.ts
index 9bcf370bd..4d895b60b 100644
--- a/packages/coding-agent/test/tools/index.test.ts
+++ b/packages/coding-agent/test/tools/index.test.ts
@@ -191,6 +191,16 @@ describe("createTools", () => {
expect(names).toContain("yield");
});
+ it("excludes todo from yield sessions unless prewalk is armed", async () => {
+ // Subagents (requireYieldTool) never get todo — except when the spawn is
+ // prewalk-armed: the prewalk plan nudge + todo gate need the child to
+ // commit its own todo list before the model hand-off.
+ const subagent = await createTools(createTestSession({ requireYieldTool: true }));
+ expect(subagent.map(t => t.name)).not.toContain("todo");
+
+ const prewalkSubagent = await createTools(createTestSession({ requireYieldTool: true, prewalkArmed: true }));
+ expect(prewalkSubagent.map(t => t.name)).toContain("todo");
+ });
it("excludes ask tool when hasUI is false", async () => {
const session = createTestSession({ hasUI: false });
diff --git a/packages/coding-agent/test/tools/schema-validation.test.ts b/packages/coding-agent/test/tools/schema-validation.test.ts
index 5f6707c19..4bfdc2ae1 100644
--- a/packages/coding-agent/test/tools/schema-validation.test.ts
+++ b/packages/coding-agent/test/tools/schema-validation.test.ts
@@ -1,5 +1,5 @@
import { describe, expect, it } from "bun:test";
-import { normalizeSchemaForGoogle, toolWireSchema } from "@oh-my-pi/pi-ai";
+import { normalizeSchemaForGoogle } from "@oh-my-pi/pi-ai";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createTools, HIDDEN_TOOLS, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
@@ -269,25 +269,6 @@ describe("tool schema validation (post-sanitization)", () => {
expect(allViolations).toEqual([]);
});
- it("bash schema and prompt advertise the timeout clamp and zero-disable", async () => {
- const session = createTestSession();
- session.settings.set("async.enabled", true);
- const tools = await createTools(session);
- const bashTool = tools.find(tool => tool.name === "bash");
- if (!bashTool?.parameters) throw new Error("bash tool parameters missing");
-
- const schema = toolWireSchema(bashTool) as {
- properties?: { timeout?: { description?: string } };
- };
- const timeoutDescription = schema.properties?.timeout?.description ?? "";
-
- expect(timeoutDescription).toContain("clamped");
- expect(timeoutDescription).toContain("1-3600");
- expect(timeoutDescription).toContain("0 disables the command deadline");
- expect(bashTool.description).toContain("nonzero values are clamped to `1..3600`");
- expect(bashTool.description).toContain("does NOT extend a nonzero timeout");
- });
-
it("hidden tools also have valid sanitized schemas", async () => {
const session = createTestSession();