diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md
index a676ecdcc..081c418f0 100644
--- a/packages/coding-agent/CHANGELOG.md
+++ b/packages/coding-agent/CHANGELOG.md
@@ -168,6 +168,11 @@
- Fixed large-session restore and `/tree` navigation blocking input while rebuilding the transcript by chunking idle rebuilds and terminal paints across event-loop turns ([#8133](https://github.com/can1357/oh-my-pi/issues/8133)).
+### Fixed
+
+- Fixed the system prompt unconditionally requiring browser verification for UI changes even when the `browser` tool is unavailable; verification now follows the actual UI surface and available tools, with a behavioral/smoke-test fallback when no runtime tool exists ([#8139](https://github.com/can1357/oh-my-pi/issues/8139)).
+- Fixed agent-facing prompts mentioning tools that may be absent from the session catalog: `todo`/`grep` workflow guidance in the system prompt, `glob`/`read`/`edit` drill-in hints in the project prompt, `ask` directives in plan mode, and the orchestrate notice's `task`/`edit`/`write`/`lsp`/`bash`/`todo` budget are now gated on tool availability.
+
## [17.2.12] - 2026-08-08
### Fixed
diff --git a/packages/coding-agent/src/modes/orchestrate.ts b/packages/coding-agent/src/modes/orchestrate.ts
index b851067a9..65750782b 100644
--- a/packages/coding-agent/src/modes/orchestrate.ts
+++ b/packages/coding-agent/src/modes/orchestrate.ts
@@ -1,3 +1,4 @@
+import { prompt } from "@oh-my-pi/pi-utils";
import orchestrateNotice from "../prompts/system/orchestrate-notice.md" with { type: "text" };
import { createGradientHighlighter, type KeywordHighlighter } from "./gradient-highlight";
import { magicKeywordRegex } from "./magic-keyword-boundary";
@@ -8,8 +9,8 @@ import { keywordInProse } from "./markdown-prose";
*
* Typing the standalone word in the input editor paints it with a cool
* teal→violet gradient ({@link highlightOrchestrate}); submitting a message that
- * mentions it appends a hidden {@link ORCHESTRATE_NOTICE} that switches the model
- * into multi-agent orchestration mode. Matching is prose-delimited and
+ * mentions it appends a hidden {@link renderOrchestrateNotice} notice that switches
+ * the model into multi-agent orchestration mode. Matching is prose-delimited and
* case-sensitive (lowercase only), so "orchestrated", "Orchestrate", or a path
* like "orchestrate.ts" never trigger either behavior. Replaces the former
* `/orchestrate` slash command.
@@ -19,7 +20,9 @@ import { keywordInProse } from "./markdown-prose";
const ORCHESTRATE_WORD = magicKeywordRegex("orchestrate");
/** Hidden system notice appended after a user message that mentions "orchestrate". */
-export const ORCHESTRATE_NOTICE: string = orchestrateNotice.trim();
+export function renderOrchestrateNotice(options: { tools: readonly string[] }): string {
+ return prompt.render(orchestrateNotice, { tools: options.tools }).trim();
+}
/**
* Whether `text` contains the standalone keyword "orchestrate" (lowercase,
diff --git a/packages/coding-agent/src/prompts/system/orchestrate-notice.md b/packages/coding-agent/src/prompts/system/orchestrate-notice.md
index 83e584fed..42cb1e01c 100644
--- a/packages/coding-agent/src/prompts/system/orchestrate-notice.md
+++ b/packages/coding-agent/src/prompts/system/orchestrate-notice.md
@@ -2,30 +2,30 @@
User message: orchestration request. Execute as orchestrator under this contract; it overrides tendencies to yield early, narrate, or do the work yourself.
-Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task` subagents. Trivial self-contained edits: make inline when dispatch overhead exceeds edit cost. Tools: planning reads; `task` dispatch; `edit`/`write` trivial inline fixes only; verification (`bun check`, `bun test`, `lsp diagnostics`); git via `bash`; `todo` tracking.
+Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task` subagents. Trivial self-contained edits: make inline when dispatch overhead exceeds edit cost. Tools: planning reads{{#has tools "task"}}; `task` dispatch{{/has}}{{#ifAny (includes tools "edit") (includes tools "write")}}; {{#has tools "edit"}}`edit`{{/has}}{{#has tools "edit"}}{{#has tools "write"}}/{{/has}}{{/has}}{{#has tools "write"}}`write`{{/has}} trivial inline fixes only{{/ifAny}}{{#ifAny (includes tools "bash") (includes tools "lsp")}}; verification ({{#has tools "bash"}}`bun check`, `bun test`{{/has}}{{#has tools "lsp"}}{{#has tools "bash"}}, {{/has}}`lsp diagnostics`{{/has}}){{/ifAny}}{{#has tools "bash"}}; git via `bash`{{/has}}{{#has tools "todo"}}; `todo` tracking{{/has}}.
1. NEVER yield before closure. Phase completion is not a yield point: launch the next phase in the same turn. Stop only when every requested item is verifiably done or concrete `[blocked]` genuinely requires the user.
-2. Before dispatch, enumerate the full surface. Expand referenced audits, plans, checklists, phase lists, and file lists into flat `todo` items. "Most"/"important" items is failure. Re-read source documents; NEVER work from memory.
+2. Before dispatch, enumerate the full surface. Expand referenced audits, plans, checklists, phase lists, and file lists into flat{{#has tools "todo"}} `todo`{{/has}} items. "Most"/"important" items is failure. Re-read source documents; NEVER work from memory.
3. Parallelize maximally; NEVER launch one-off `task`. Disjoint-scope edits MUST be parallel `task` calls in one message. Divisible work: split and dispatch together, never serially. Before exactly one subagent: find parallel work and dispatch it, or make the small change inline. Serialize only when a produced contract—types, schema, shared module—is consumed next; state the dependency.
4. Every `task` self-contained; subagents share no context. Specify ≤3–5 explicit target paths (no globs), change APIs/patterns, edge cases, observable acceptance criteria. NEVER assume a shared plan.
-5. Verify each phase before the next: `bun check` types, package-scoped `bun test` behavior, `lsp diagnostics` changed files. Breakage: dispatch fix-up subagents, then re-verify before advancing. NEVER declare a red tree done.
+5. Verify each phase before the next{{#ifAny (includes tools "bash") (includes tools "lsp")}}: {{#has tools "bash"}}`bun check` types, package-scoped `bun test` behavior{{/has}}{{#has tools "lsp"}}{{#has tools "bash"}}, {{/has}}`lsp diagnostics` changed files{{/has}}{{/ifAny}}. Breakage: dispatch fix-up subagents, then re-verify before advancing. NEVER declare a red tree done.
6. Commit only if requested or repo workflow expects it: after each green phase, focused phase-naming message. NEVER commit red trees or unrequested work.
7. Incomplete/wrong subagent work: spawn corrective subagent specifying the gap; NEVER silently fix it inline.
8. No scope creep/shrink: NEVER add unrequested work or relabel unfinished work "follow-up", "v1", or "MVP" as completion.
9. Subagents NEVER verify, lint, or format. Every `task` MUST say to skip gates/formatters; edit only. At phase end, orchestrator verifies and formats once across the union of changed files, avoiding redundant/racing formatter runs.
-10. Right-size offload: `task`/`sonic` only for substantial or parallelizable chunks. Trivial self-contained mechanical edits—delete one redundant glob, fix one config line, rename one symbol in one file—make inline with `edit`/`write`; dispatch costs more than Goal/Constraints description.
+10. Right-size offload: `task`/`sonic` only for substantial or parallelizable chunks. Trivial self-contained mechanical edits—delete one redundant glob, fix one config line, rename one symbol in one file—make inline{{#ifAny (includes tools "edit") (includes tools "write")}} with {{#has tools "edit"}}`edit`{{/has}}{{#has tools "edit"}}{{#has tools "write"}}/{{/has}}{{/has}}{{#has tools "write"}}`write`{{/has}}{{/ifAny}}; dispatch costs more than Goal/Constraints description.
1. Ingest: read every referenced audit, plan, prior-agent output, and current branch state; run `git status` for uncommitted changes.
-2. Plan: materialize full work surface in ordered `todo` phases; list each phase's parallel units.
+2. Plan: materialize full work surface{{#has tools "todo"}} in ordered `todo` phases{{/has}}; list each phase's parallel units.
3. Dispatch: launch all parallel `task` subagents in one message; collect every result (async results / `hub` wait) before advancing.
4. Verify: run gates; on failure dispatch fix-ups and re-verify. Never advance on red.
5. Commit if applicable: focused phase-naming message.
-6. Advance: mark phase done in `todo`; immediately start next. No inter-phase summary.
-7. Final verification: after last green phase, rerun full gates; confirm every `todo` closed; yield terse status, not recap.
+6. Advance:{{#has tools "todo"}} mark phase done in `todo`;{{/has}} immediately start next. No inter-phase summary.
+7. Final verification: after last green phase, rerun full gates; confirm every{{#has tools "todo"}} `todo`{{/has}} item closed; yield terse status, not recap.
@@ -34,7 +34,7 @@ Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task`
- Yielding after phase 1 with "ready to continue?".
- Serial subagent dispatch when five can run in parallel.
- Skipping between-phase `bun check` because change "looked safe".
-- Closing todos from subagent reports without gate verification.
-- Chat progress summaries instead of advancing.
+- {{#has tools "todo"}}Closing todos from subagent reports without gate verification.
+{{/has}}- Chat progress summaries instead of advancing.
diff --git a/packages/coding-agent/src/prompts/system/plan-mode-active.md b/packages/coding-agent/src/prompts/system/plan-mode-active.md
index a5ac17e45..6ebf8b1ab 100644
--- a/packages/coding-agent/src/prompts/system/plan-mode-active.md
+++ b/packages/coding-agent/src/prompts/system/plan-mode-active.md
@@ -38,8 +38,8 @@ Write each section with body: `N*` requires multiline section; bare heading →
Resolve unknowns by discovery, not questions.
-- Discoverable facts — locations, behavior, signatures, configs: MUST discover with `glob`, `grep`, `read`,{{#if scoutAvailable}} or parallel `scout` subagents{{/if}}. Every asserted path, symbol, signature, behavior: actually read this session. Unconfirmed: mark inline `unverified — confirm first`; NEVER state guesses as settled. Ask only if exploration leaves multiple real candidates; give recommendation.
-- Preferences/tradeoffs — intent, UX, scope edges, performance vs. simplicity: not code-derivable. Ask early via `{{askToolName}}`: 2–4 mutually exclusive options + recommended default. Unanswered → use default; record under Assumptions.
+- Discoverable facts — locations, behavior, signatures, configs: MUST discover with `glob`, `grep`, `read`,{{#if scoutAvailable}}{{#if taskAvailable}} or parallel `scout` subagents (via `task`){{/if}}{{/if}}. Every asserted path, symbol, signature, behavior: actually read this session. Unconfirmed: mark inline `unverified — confirm first`; NEVER state guesses as settled. Ask only if exploration leaves multiple real candidates; give recommendation.
+- Preferences/tradeoffs — intent, UX, scope edges, performance vs. simplicity: not code-derivable.{{#if askAvailable}} Ask early via `{{askToolName}}`: 2–4 mutually exclusive options + recommended default.{{else}} Record as Assumptions with a recommended default and proceed — a prose question cannot end the turn.{{/if}} Unanswered → use default; record under Assumptions.
Every question MUST alter plan or resolve load-bearing choice; batch. NEVER ask what exploration answers or filler.
@@ -62,7 +62,7 @@ New request primary; existing plan reference only. NEVER reconcile old plan whil
1. **Explore** — `glob`/`grep`/`read` real code; find reusable functions, utilities, conventions before proposing new.
-2. **Interview** — `{{askToolName}}` only for preferences/tradeoffs; batch; NEVER ask what exploration answers.
+2. **Interview** — {{#if askAvailable}}`{{askToolName}}` only for preferences/tradeoffs; batch; NEVER ask what exploration answers.{{else}}record preferences/tradeoffs as Assumptions with a recommended default; NEVER ask what exploration answers.{{/if}}
3. **Update** — revise plan with `{{editToolName}}` while learning.
4. **Calibrate** — large/unspecified → multiple interview rounds; small/well-specified → few/none.
@@ -70,9 +70,9 @@ New request primary; existing plan reference only. NEVER reconcile old plan whil
## Workflow — parallel
-1. **Understand** — request and supporting code.{{#if scoutAvailable}} Scope spans areas → parallel `scout` subagents via `task`, distinct focuses: implementations, related components, test patterns.{{/if}} Find reusable code before proposing new.
+1. **Understand** — request and supporting code.{{#if scoutAvailable}}{{#if taskAvailable}} Scope spans areas → parallel `scout` subagents via `task`, distinct focuses: implementations, related components, test patterns.{{/if}}{{/if}} Find reusable code before proposing new.
2. **Design** — draft approach from findings, briefly weigh tradeoffs, commit. Large/cross-cutting → MAY spawn critique subagent before commitment.
-3. **Review** — read intended files; validate approach against code and literal request; `{{askToolName}}` resolves remaining preferences.
+3. **Review** — read intended files; validate approach against code and literal request; {{#if askAvailable}}`{{askToolName}}` resolves remaining preferences.{{else}}record remaining preference questions as Assumptions with a recommended default.{{/if}}
4. **Write** — plan per **Plan contents**.
{{/if}}
@@ -115,8 +115,8 @@ All require self-contained file.
Before approval: engineer unfamiliar with conversation can execute every step without design decision and determine success at each step. Otherwise deepen any choice-forcing or ambiguous-done step.
Turn ends ONLY:
-1. `{{askToolName}}` gathers requirements/chooses approaches; OR
+1. {{#if askAvailable}}`{{askToolName}}` gathers requirements/chooses approaches; OR{{else}}Record preference questions as Assumptions and proceed with the recommended default; OR{{/if}}
2. `{{writeToolName}}` writes plan ``/title as plain text to `xd://propose` (`local://-plan.md` slug).
-NEVER request plan approval via prose/`{{askToolName}}`; MUST use `xd://propose` write. MUST continue until decision-complete.
+NEVER request plan approval via prose/{{#if askAvailable}}`{{askToolName}}`{{else}}a question{{/if}}; MUST use `xd://propose` write. MUST continue until decision-complete.
diff --git a/packages/coding-agent/src/prompts/system/project-prompt.md b/packages/coding-agent/src/prompts/system/project-prompt.md
index e8b7b91f1..a35e4a0c0 100644
--- a/packages/coding-agent/src/prompts/system/project-prompt.md
+++ b/packages/coding-agent/src/prompts/system/project-prompt.md
@@ -34,14 +34,14 @@ Context files above auto-loaded. NEVER `grep`/`glob` for `AGENTS.md`, `CLAUDE.md
Working-directory layout: newest mtime first; depth ≤ 3.
{{workspaceTree.rendered}}
{{#if workspaceTree.truncated}}
-Some entries elided to shorten tree — use `glob`/`read` to drill in.
+{{#has tools "glob"}}{{#has tools "read"}}Some entries elided to shorten tree — use `{{toolRefs.glob}}`/`{{toolRefs.read}}` to drill in.{{/has}}{{/has}}
{{/if}}
{{/if}}
{{/if}}
{{#if additionalWorkspaceRoots.length}}
-Additional workspace directories. This CURRENT workspace state supersedes workspace changes mentioned earlier in the conversation. Use absolute paths under these roots to `read`/`grep`/`glob`/`edit`. Manage with `/add-dir` and `/remove-dir`; `/dirs` lists them.
+Additional workspace directories. This CURRENT workspace state supersedes workspace changes mentioned earlier in the conversation. {{#ifAny (includes tools "read") (includes tools "grep") (includes tools "glob") (includes tools "edit")}}Use absolute paths under these roots to {{#has tools "read"}}`{{toolRefs.read}}`{{/has}}{{#has tools "grep"}}{{#ifAny (includes tools "read")}}/{{/ifAny}}`{{toolRefs.grep}}`{{/has}}{{#has tools "glob"}}{{#ifAny (includes tools "read") (includes tools "grep")}}/{{/ifAny}}`{{toolRefs.glob}}`{{/has}}{{#has tools "edit"}}{{#ifAny (includes tools "read") (includes tools "grep") (includes tools "glob")}}/{{/ifAny}}`{{toolRefs.edit}}`{{/has}}.{{/ifAny}} Manage with `/add-dir` and `/remove-dir`; `/dirs` lists them.
{{#each additionalWorkspaceRoots}}
- {{this}}
{{/each}}
diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md
index 58fe60192..ad26d0f2b 100644
--- a/packages/coding-agent/src/prompts/system/system-prompt.md
+++ b/packages/coding-agent/src/prompts/system/system-prompt.md
@@ -124,9 +124,11 @@ MUST use specialized tool over shell equivalent:
{{#has tools "bash"}}- Bash litmus: one external-CLI call/short pipeline returning count, frequency, set difference, checksum. For merely moving, paging, trimming fetchable bytes: tool.{{/has}}
{{#if autoQaEnabled}}
+{{#has tools "write"}}
`{{toolRefs.write}} xd://report_issue`: automated QA. Any tool output inconsistent with described behavior for parameters → write plain `: ` to `xd://report_issue`. False positives fine.
+{{/has}}
{{/if}}
# Exploration
@@ -179,8 +181,9 @@ Delegation preferred. Once design settles, SHOULD fan substantial work to `{{too
- Tool failure/file change since read → re-read before acting.
# 3. Decompose
-- Update todos; skip trivial requests.
+{{#has tools "todo"}}- Update todos; skip trivial requests.
- Todo calls NEVER alone: batch each with turn's real calls (`init` with first reads/edits; `done` with next action/final verification). Todo-only assistant turn wastes round trip.
+{{/has}}
# 4. Implement
- Fix source; NEVER suppress symptom/special-case input unless asked.
@@ -191,7 +194,17 @@ Delegation preferred. Once design settles, SHOULD fan substantial work to `{{too
# 5. Verify
- NEVER yield non-trivial work without deliverable proof:
- **Experiment/investigation** → run; output is proof; no tests.
- - **UI change** → browser-drive; visual confirmation is proof; no tests unless existing suite really breaks.
+ - **UI change** → verify against the actual surface:
+{{#has tools "browser"}}
+ - **Web UI** → browser-drive with `{{toolRefs.browser}}`; visual confirmation is proof; no tests unless existing suite really breaks.
+{{/has}}
+{{#has tools "computer"}}
+ - **Native desktop UI** → drive with `{{toolRefs.computer}}`; ground every claim in fresh screenshot or accessibility evidence.
+{{/has}}
+ - **TUI/CLI** → launch the actual program and verify terminal interaction, output, or state.
+{{#ifAny (not (includes tools "browser")) (not (includes tools "computer"))}}
+ - No suitable runtime tool for the changed surface → verify with a behavioral test or smoke test; explicitly report when visual verification cannot be performed.
+{{/ifAny}}
- **Bug fix** → reproduce, fix, confirm reproduction no longer triggers.
- **Permanent feature/API change** → existing changed-contract tests. Add test only for uncovered new observable contract or user request.
- Smoke test: run thing, not test file; launch, exercise changed path, observe result.
diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts
index 4756dd5c6..61f851008 100644
--- a/packages/coding-agent/src/session/agent-session.ts
+++ b/packages/coding-agent/src/session/agent-session.ts
@@ -148,7 +148,7 @@ import type { IrcMessage } from "../irc/bus";
import type { DaemonCompletionNotification } from "../launch/protocol";
import { shutdownMnemopiEmbedClient } from "../mnemopi/embed-client";
import { getMnemopiSessionState, type MnemopiSessionState, setMnemopiSessionState } from "../mnemopi/state";
-import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate";
+import { containsOrchestrate, renderOrchestrateNotice } from "../modes/orchestrate";
import { theme } from "../modes/theme/theme";
import { parseTurnBudget } from "../modes/turn-budget";
import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink";
@@ -4955,12 +4955,15 @@ export class AgentSession {
: sessionPlanUrl;
const planExists = fs.existsSync(resolvedPlanPath);
+ const activeToolNames = this.getActiveToolNames();
const content = prompt.render(planModeActivePrompt, {
planFilePath: displayPlanPath,
planExists,
askToolName: "ask",
writeToolName: "write",
editToolName: "edit",
+ askAvailable: activeToolNames.includes("ask"),
+ taskAvailable: activeToolNames.includes("task"),
isHashlineEditMode: this.#resolveActiveEditMode() === "hashline",
reentry: state.reentry ?? false,
iterative: state.workflow === "iterative",
@@ -5081,14 +5084,19 @@ export class AgentSession {
});
}
if (this.#magicKeywordEnabled("orchestrate") && containsOrchestrate(text)) {
- keywordNotices.push({
- role: "custom",
- customType: "orchestrate-notice",
- content: ORCHESTRATE_NOTICE,
- display: false,
- attribution: "user",
- timestamp,
- });
+ const activeToolNames = this.getActiveToolNames();
+ // The contract is entirely about `task` subagent dispatch; without the
+ // task tool the notice would demand an unavailable capability.
+ if (activeToolNames.includes("task")) {
+ keywordNotices.push({
+ role: "custom",
+ customType: "orchestrate-notice",
+ content: renderOrchestrateNotice({ tools: activeToolNames }),
+ display: false,
+ attribution: "user",
+ timestamp,
+ });
+ }
}
if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) {
const activeToolNames = this.getActiveToolNames();
diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts
index 86e62affc..e38fc8fcb 100644
--- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts
+++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts
@@ -171,6 +171,18 @@ describe("AgentSession magic keyword settings", () => {
expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]);
});
+ it("skips orchestrate notice when the task tool is inactive", async () => {
+ const created = await createMagicKeywordSession(root, []);
+ session = created.session;
+ authStorage = created.authStorage;
+ const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
+
+ await session.prompt("please orchestrate this");
+
+ const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ customType?: string }>;
+ expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]);
+ });
+
it("skips workflowz notice when the eval tool is inactive", async () => {
const created = await createMagicKeywordSession(root, [mockTaskTool]);
session = created.session;
diff --git a/packages/coding-agent/test/modes/orchestrate.test.ts b/packages/coding-agent/test/modes/orchestrate.test.ts
index c1cac442a..e5ad21e27 100644
--- a/packages/coding-agent/test/modes/orchestrate.test.ts
+++ b/packages/coding-agent/test/modes/orchestrate.test.ts
@@ -2,7 +2,7 @@ import { beforeAll, describe, expect, it } from "bun:test";
import {
containsOrchestrate,
highlightOrchestrate,
- ORCHESTRATE_NOTICE,
+ renderOrchestrateNotice,
} from "@oh-my-pi/pi-coding-agent/modes/orchestrate";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { containsUltrathink, highlightUltrathink } from "@oh-my-pi/pi-coding-agent/modes/ultrathink";
@@ -85,11 +85,37 @@ describe("orchestrate keyword highlighting", () => {
describe("orchestrate notice", () => {
it("is a self-contained system notice carrying the orchestration contract", () => {
- expect(ORCHESTRATE_NOTICE.startsWith("")).toBe(true);
- expect(ORCHESTRATE_NOTICE.endsWith("")).toBe(true);
- expect(ORCHESTRATE_NOTICE).toContain("orchestrator");
+ const notice = renderOrchestrateNotice({
+ tools: ["read", "task", "edit", "write", "lsp", "bash", "todo"],
+ });
+ expect(notice.startsWith("")).toBe(true);
+ expect(notice.endsWith("")).toBe(true);
+ expect(notice).toContain("orchestrator");
// The contract must not retain the slash-command input placeholder.
- expect(ORCHESTRATE_NOTICE).not.toContain("$@");
+ expect(notice).not.toContain("$@");
+ });
+
+ it("omits tool-budget mentions for tools absent from the session", () => {
+ const notice = renderOrchestrateNotice({ tools: ["read"] });
+ expect(notice).not.toContain("`task` for dispatch");
+ expect(notice).not.toContain("`edit`");
+ expect(notice).not.toContain("`write`");
+ expect(notice).not.toContain("`lsp diagnostics`");
+ expect(notice).not.toContain("via `bash`");
+ expect(notice).not.toContain("`todo` for tracking");
+ });
+
+ it("does not name edit when only write is available", () => {
+ const writeOnly = renderOrchestrateNotice({ tools: ["read", "write"] });
+ expect(writeOnly).toContain("with `write`");
+ expect(writeOnly).not.toContain("`edit`/`write`");
+ expect(writeOnly).not.toContain("with `edit`");
+ });
+
+ it("does not name write when only edit is available", () => {
+ const editOnly = renderOrchestrateNotice({ tools: ["read", "edit"] });
+ expect(editOnly).toContain("with `edit`");
+ expect(editOnly).not.toContain("`edit`/`write`");
});
});
diff --git a/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts b/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts
index 00cfffb4e..f2313d0ab 100644
--- a/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts
+++ b/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts
@@ -9,15 +9,57 @@ const BASE = {
editToolName: "edit",
isHashlineEditMode: false,
iterative: false,
+ askAvailable: true,
+ taskAvailable: true,
+ scoutAvailable: true,
+ reentry: false,
+ planExists: true,
} as const;
-function render(overrides: { reentry: boolean; planExists: boolean }): string {
+type Overrides = Partial>;
+
+function render(overrides: Overrides = {}): string {
return prompt.render(planModeActivePrompt, { ...BASE, ...overrides });
}
describe("plan-mode re-entry prompt", () => {
it("only emits the Re-entry section when re-entering", () => {
- expect(render({ reentry: false, planExists: true })).not.toContain("## Re-entry");
- expect(render({ reentry: true, planExists: true })).toContain("## Re-entry");
+ expect(render({ reentry: false })).not.toContain("## Re-entry");
+ expect(render({ reentry: true })).toContain("## Re-entry");
+ });
+});
+
+describe("plan-mode-active tool availability", () => {
+ it("omits ask-tool directives when ask is unavailable", () => {
+ const withoutAsk = render({ askAvailable: false, iterative: true });
+ expect(withoutAsk).not.toContain("`ask` with 2–4 mutually exclusive options");
+ expect(withoutAsk).not.toContain("use `ask` for preferences and tradeoffs");
+ expect(withoutAsk).not.toContain("Using `ask` to gather requirements");
+
+ const withAsk = render({ askAvailable: true, iterative: true });
+ expect(withAsk).toContain("`ask` with 2–4 mutually exclusive options");
+ expect(withAsk).toContain("use `ask` for preferences and tradeoffs");
+ });
+
+ it("records preferences as assumptions when ask is unavailable", () => {
+ const iterativeWithoutAsk = render({ askAvailable: false, iterative: true });
+ expect(iterativeWithoutAsk).toContain("Record them as Assumptions with a recommended default");
+ expect(iterativeWithoutAsk).toContain("record preferences and tradeoffs as Assumptions");
+ expect(iterativeWithoutAsk).not.toContain("`ask` for preferences and tradeoffs only");
+
+ const parallelWithoutAsk = render({ askAvailable: false, iterative: false });
+ expect(parallelWithoutAsk).toContain(
+ "record any remaining preference questions as Assumptions with a recommended default",
+ );
+ // A prose question cannot end the turn in plan mode — no prose-terminal option.
+ expect(parallelWithoutAsk).not.toContain("Presenting a choice between approaches");
+ });
+
+ it("omits scout-via-task dispatch when the task tool is unavailable", () => {
+ const withoutTask = render({ taskAvailable: false, scoutAvailable: true });
+ expect(withoutTask).not.toContain("(via `task`)");
+
+ const withTask = render({ taskAvailable: true, scoutAvailable: true });
+ expect(withTask).toContain("(via `task`)");
});
});
diff --git a/packages/coding-agent/test/system-prompt-inventory.test.ts b/packages/coding-agent/test/system-prompt-inventory.test.ts
index 67933b61a..ba2ededd4 100644
--- a/packages/coding-agent/test/system-prompt-inventory.test.ts
+++ b/packages/coding-agent/test/system-prompt-inventory.test.ts
@@ -682,4 +682,70 @@ describe("system prompt tool inventory", () => {
expect(withScout).toContain("one read-only scout while working is allowed");
expect(withoutScout).not.toContain("read-only scout");
});
+
+ it("does not require browser verification when the browser tool is absent (issue #8139)", async () => {
+ const opts = {
+ cwd: tempDir,
+ contextFiles: [],
+ skills: [],
+ rules: [],
+ workspaceTree: { ...EMPTY_TREE, rootPath: tempDir },
+ };
+ const tools = new Map(TOOLS);
+ const withoutBrowser = (
+ await buildSystemPrompt({
+ ...opts,
+ toolNames: ["read", "bash"],
+ tools,
+ nativeTools: true,
+ inlineToolDescriptors: false,
+ })
+ ).systemPrompt.join("\n\n");
+
+ expect(withoutBrowser).not.toContain("drive it in `browser`");
+ expect(withoutBrowser).not.toContain("drive it in browser");
+ expect(withoutBrowser).toContain("TUI/CLI");
+ expect(withoutBrowser).toContain("behavioral test or smoke test");
+
+ tools.set("browser", {
+ label: "Browser",
+ description: "Drives a real Chromium tab.",
+ parameters: { type: "object", properties: {} },
+ });
+ const withBrowser = (
+ await buildSystemPrompt({
+ ...opts,
+ toolNames: ["read", "bash", "browser"],
+ tools,
+ nativeTools: true,
+ inlineToolDescriptors: false,
+ })
+ ).systemPrompt.join("\n\n");
+
+ expect(withBrowser).toContain("drive it in `browser`");
+ // A browser-only session still needs the smoke-test fallback for
+ // native-desktop surfaces (no computer tool).
+ expect(withBrowser).toContain("behavioral test or smoke test");
+ });
+
+ it("omits todo workflow guidance when the todo tool is absent", async () => {
+ const opts = {
+ cwd: tempDir,
+ contextFiles: [],
+ skills: [],
+ rules: [],
+ workspaceTree: { ...EMPTY_TREE, rootPath: tempDir },
+ tools: TOOLS,
+ nativeTools: true,
+ inlineToolDescriptors: false,
+ };
+ const withoutTodo = (await buildSystemPrompt({ ...opts, toolNames: ["read", "bash"] })).systemPrompt.join("\n\n");
+ expect(withoutTodo).not.toContain("Todo calls NEVER travel alone");
+ expect(withoutTodo).not.toContain("batch every todo op");
+
+ const withTodo = (await buildSystemPrompt({ ...opts, toolNames: ["read", "bash", "todo"] })).systemPrompt.join(
+ "\n\n",
+ );
+ expect(withTodo).toContain("Todo calls NEVER travel alone");
+ });
});