diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a676ecdcc..081c418f0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -168,6 +168,11 @@ - Fixed large-session restore and `/tree` navigation blocking input while rebuilding the transcript by chunking idle rebuilds and terminal paints across event-loop turns ([#8133](https://github.com/can1357/oh-my-pi/issues/8133)). +### Fixed + +- Fixed the system prompt unconditionally requiring browser verification for UI changes even when the `browser` tool is unavailable; verification now follows the actual UI surface and available tools, with a behavioral/smoke-test fallback when no runtime tool exists ([#8139](https://github.com/can1357/oh-my-pi/issues/8139)). +- Fixed agent-facing prompts mentioning tools that may be absent from the session catalog: `todo`/`grep` workflow guidance in the system prompt, `glob`/`read`/`edit` drill-in hints in the project prompt, `ask` directives in plan mode, and the orchestrate notice's `task`/`edit`/`write`/`lsp`/`bash`/`todo` budget are now gated on tool availability. + ## [17.2.12] - 2026-08-08 ### Fixed diff --git a/packages/coding-agent/src/modes/orchestrate.ts b/packages/coding-agent/src/modes/orchestrate.ts index b851067a9..65750782b 100644 --- a/packages/coding-agent/src/modes/orchestrate.ts +++ b/packages/coding-agent/src/modes/orchestrate.ts @@ -1,3 +1,4 @@ +import { prompt } from "@oh-my-pi/pi-utils"; import orchestrateNotice from "../prompts/system/orchestrate-notice.md" with { type: "text" }; import { createGradientHighlighter, type KeywordHighlighter } from "./gradient-highlight"; import { magicKeywordRegex } from "./magic-keyword-boundary"; @@ -8,8 +9,8 @@ import { keywordInProse } from "./markdown-prose"; * * Typing the standalone word in the input editor paints it with a cool * teal→violet gradient ({@link highlightOrchestrate}); submitting a message that - * mentions it appends a hidden {@link ORCHESTRATE_NOTICE} that switches the model - * into multi-agent orchestration mode. Matching is prose-delimited and + * mentions it appends a hidden {@link renderOrchestrateNotice} notice that switches + * the model into multi-agent orchestration mode. Matching is prose-delimited and * case-sensitive (lowercase only), so "orchestrated", "Orchestrate", or a path * like "orchestrate.ts" never trigger either behavior. Replaces the former * `/orchestrate` slash command. @@ -19,7 +20,9 @@ import { keywordInProse } from "./markdown-prose"; const ORCHESTRATE_WORD = magicKeywordRegex("orchestrate"); /** Hidden system notice appended after a user message that mentions "orchestrate". */ -export const ORCHESTRATE_NOTICE: string = orchestrateNotice.trim(); +export function renderOrchestrateNotice(options: { tools: readonly string[] }): string { + return prompt.render(orchestrateNotice, { tools: options.tools }).trim(); +} /** * Whether `text` contains the standalone keyword "orchestrate" (lowercase, diff --git a/packages/coding-agent/src/prompts/system/orchestrate-notice.md b/packages/coding-agent/src/prompts/system/orchestrate-notice.md index 83e584fed..42cb1e01c 100644 --- a/packages/coding-agent/src/prompts/system/orchestrate-notice.md +++ b/packages/coding-agent/src/prompts/system/orchestrate-notice.md @@ -2,30 +2,30 @@ User message: orchestration request. Execute as orchestrator under this contract; it overrides tendencies to yield early, narrate, or do the work yourself. -Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task` subagents. Trivial self-contained edits: make inline when dispatch overhead exceeds edit cost. Tools: planning reads; `task` dispatch; `edit`/`write` trivial inline fixes only; verification (`bun check`, `bun test`, `lsp diagnostics`); git via `bash`; `todo` tracking. +Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task` subagents. Trivial self-contained edits: make inline when dispatch overhead exceeds edit cost. Tools: planning reads{{#has tools "task"}}; `task` dispatch{{/has}}{{#ifAny (includes tools "edit") (includes tools "write")}}; {{#has tools "edit"}}`edit`{{/has}}{{#has tools "edit"}}{{#has tools "write"}}/{{/has}}{{/has}}{{#has tools "write"}}`write`{{/has}} trivial inline fixes only{{/ifAny}}{{#ifAny (includes tools "bash") (includes tools "lsp")}}; verification ({{#has tools "bash"}}`bun check`, `bun test`{{/has}}{{#has tools "lsp"}}{{#has tools "bash"}}, {{/has}}`lsp diagnostics`{{/has}}){{/ifAny}}{{#has tools "bash"}}; git via `bash`{{/has}}{{#has tools "todo"}}; `todo` tracking{{/has}}. 1. NEVER yield before closure. Phase completion is not a yield point: launch the next phase in the same turn. Stop only when every requested item is verifiably done or concrete `[blocked]` genuinely requires the user. -2. Before dispatch, enumerate the full surface. Expand referenced audits, plans, checklists, phase lists, and file lists into flat `todo` items. "Most"/"important" items is failure. Re-read source documents; NEVER work from memory. +2. Before dispatch, enumerate the full surface. Expand referenced audits, plans, checklists, phase lists, and file lists into flat{{#has tools "todo"}} `todo`{{/has}} items. "Most"/"important" items is failure. Re-read source documents; NEVER work from memory. 3. Parallelize maximally; NEVER launch one-off `task`. Disjoint-scope edits MUST be parallel `task` calls in one message. Divisible work: split and dispatch together, never serially. Before exactly one subagent: find parallel work and dispatch it, or make the small change inline. Serialize only when a produced contract—types, schema, shared module—is consumed next; state the dependency. 4. Every `task` self-contained; subagents share no context. Specify ≤3–5 explicit target paths (no globs), change APIs/patterns, edge cases, observable acceptance criteria. NEVER assume a shared plan. -5. Verify each phase before the next: `bun check` types, package-scoped `bun test` behavior, `lsp diagnostics` changed files. Breakage: dispatch fix-up subagents, then re-verify before advancing. NEVER declare a red tree done. +5. Verify each phase before the next{{#ifAny (includes tools "bash") (includes tools "lsp")}}: {{#has tools "bash"}}`bun check` types, package-scoped `bun test` behavior{{/has}}{{#has tools "lsp"}}{{#has tools "bash"}}, {{/has}}`lsp diagnostics` changed files{{/has}}{{/ifAny}}. Breakage: dispatch fix-up subagents, then re-verify before advancing. NEVER declare a red tree done. 6. Commit only if requested or repo workflow expects it: after each green phase, focused phase-naming message. NEVER commit red trees or unrequested work. 7. Incomplete/wrong subagent work: spawn corrective subagent specifying the gap; NEVER silently fix it inline. 8. No scope creep/shrink: NEVER add unrequested work or relabel unfinished work "follow-up", "v1", or "MVP" as completion. 9. Subagents NEVER verify, lint, or format. Every `task` MUST say to skip gates/formatters; edit only. At phase end, orchestrator verifies and formats once across the union of changed files, avoiding redundant/racing formatter runs. -10. Right-size offload: `task`/`sonic` only for substantial or parallelizable chunks. Trivial self-contained mechanical edits—delete one redundant glob, fix one config line, rename one symbol in one file—make inline with `edit`/`write`; dispatch costs more than Goal/Constraints description. +10. Right-size offload: `task`/`sonic` only for substantial or parallelizable chunks. Trivial self-contained mechanical edits—delete one redundant glob, fix one config line, rename one symbol in one file—make inline{{#ifAny (includes tools "edit") (includes tools "write")}} with {{#has tools "edit"}}`edit`{{/has}}{{#has tools "edit"}}{{#has tools "write"}}/{{/has}}{{/has}}{{#has tools "write"}}`write`{{/has}}{{/ifAny}}; dispatch costs more than Goal/Constraints description. 1. Ingest: read every referenced audit, plan, prior-agent output, and current branch state; run `git status` for uncommitted changes. -2. Plan: materialize full work surface in ordered `todo` phases; list each phase's parallel units. +2. Plan: materialize full work surface{{#has tools "todo"}} in ordered `todo` phases{{/has}}; list each phase's parallel units. 3. Dispatch: launch all parallel `task` subagents in one message; collect every result (async results / `hub` wait) before advancing. 4. Verify: run gates; on failure dispatch fix-ups and re-verify. Never advance on red. 5. Commit if applicable: focused phase-naming message. -6. Advance: mark phase done in `todo`; immediately start next. No inter-phase summary. -7. Final verification: after last green phase, rerun full gates; confirm every `todo` closed; yield terse status, not recap. +6. Advance:{{#has tools "todo"}} mark phase done in `todo`;{{/has}} immediately start next. No inter-phase summary. +7. Final verification: after last green phase, rerun full gates; confirm every{{#has tools "todo"}} `todo`{{/has}} item closed; yield terse status, not recap. @@ -34,7 +34,7 @@ Decompose, dispatch, verify, iterate. Substantial or parallelizable work: `task` - Yielding after phase 1 with "ready to continue?". - Serial subagent dispatch when five can run in parallel. - Skipping between-phase `bun check` because change "looked safe". -- Closing todos from subagent reports without gate verification. -- Chat progress summaries instead of advancing. +- {{#has tools "todo"}}Closing todos from subagent reports without gate verification. +{{/has}}- Chat progress summaries instead of advancing. diff --git a/packages/coding-agent/src/prompts/system/plan-mode-active.md b/packages/coding-agent/src/prompts/system/plan-mode-active.md index a5ac17e45..6ebf8b1ab 100644 --- a/packages/coding-agent/src/prompts/system/plan-mode-active.md +++ b/packages/coding-agent/src/prompts/system/plan-mode-active.md @@ -38,8 +38,8 @@ Write each section with body: `N*` requires multiline section; bare heading → Resolve unknowns by discovery, not questions. -- Discoverable facts — locations, behavior, signatures, configs: MUST discover with `glob`, `grep`, `read`,{{#if scoutAvailable}} or parallel `scout` subagents{{/if}}. Every asserted path, symbol, signature, behavior: actually read this session. Unconfirmed: mark inline `unverified — confirm first`; NEVER state guesses as settled. Ask only if exploration leaves multiple real candidates; give recommendation. -- Preferences/tradeoffs — intent, UX, scope edges, performance vs. simplicity: not code-derivable. Ask early via `{{askToolName}}`: 2–4 mutually exclusive options + recommended default. Unanswered → use default; record under Assumptions. +- Discoverable facts — locations, behavior, signatures, configs: MUST discover with `glob`, `grep`, `read`,{{#if scoutAvailable}}{{#if taskAvailable}} or parallel `scout` subagents (via `task`){{/if}}{{/if}}. Every asserted path, symbol, signature, behavior: actually read this session. Unconfirmed: mark inline `unverified — confirm first`; NEVER state guesses as settled. Ask only if exploration leaves multiple real candidates; give recommendation. +- Preferences/tradeoffs — intent, UX, scope edges, performance vs. simplicity: not code-derivable.{{#if askAvailable}} Ask early via `{{askToolName}}`: 2–4 mutually exclusive options + recommended default.{{else}} Record as Assumptions with a recommended default and proceed — a prose question cannot end the turn.{{/if}} Unanswered → use default; record under Assumptions. Every question MUST alter plan or resolve load-bearing choice; batch. NEVER ask what exploration answers or filler. @@ -62,7 +62,7 @@ New request primary; existing plan reference only. NEVER reconcile old plan whil 1. **Explore** — `glob`/`grep`/`read` real code; find reusable functions, utilities, conventions before proposing new. -2. **Interview** — `{{askToolName}}` only for preferences/tradeoffs; batch; NEVER ask what exploration answers. +2. **Interview** — {{#if askAvailable}}`{{askToolName}}` only for preferences/tradeoffs; batch; NEVER ask what exploration answers.{{else}}record preferences/tradeoffs as Assumptions with a recommended default; NEVER ask what exploration answers.{{/if}} 3. **Update** — revise plan with `{{editToolName}}` while learning. 4. **Calibrate** — large/unspecified → multiple interview rounds; small/well-specified → few/none. @@ -70,9 +70,9 @@ New request primary; existing plan reference only. NEVER reconcile old plan whil ## Workflow — parallel -1. **Understand** — request and supporting code.{{#if scoutAvailable}} Scope spans areas → parallel `scout` subagents via `task`, distinct focuses: implementations, related components, test patterns.{{/if}} Find reusable code before proposing new. +1. **Understand** — request and supporting code.{{#if scoutAvailable}}{{#if taskAvailable}} Scope spans areas → parallel `scout` subagents via `task`, distinct focuses: implementations, related components, test patterns.{{/if}}{{/if}} Find reusable code before proposing new. 2. **Design** — draft approach from findings, briefly weigh tradeoffs, commit. Large/cross-cutting → MAY spawn critique subagent before commitment. -3. **Review** — read intended files; validate approach against code and literal request; `{{askToolName}}` resolves remaining preferences. +3. **Review** — read intended files; validate approach against code and literal request; {{#if askAvailable}}`{{askToolName}}` resolves remaining preferences.{{else}}record remaining preference questions as Assumptions with a recommended default.{{/if}} 4. **Write** — plan per **Plan contents**. {{/if}} @@ -115,8 +115,8 @@ All require self-contained file. Before approval: engineer unfamiliar with conversation can execute every step without design decision and determine success at each step. Otherwise deepen any choice-forcing or ambiguous-done step. Turn ends ONLY: -1. `{{askToolName}}` gathers requirements/chooses approaches; OR +1. {{#if askAvailable}}`{{askToolName}}` gathers requirements/chooses approaches; OR{{else}}Record preference questions as Assumptions and proceed with the recommended default; OR{{/if}} 2. `{{writeToolName}}` writes plan ``/title as plain text to `xd://propose` (`local://-plan.md` slug). -NEVER request plan approval via prose/`{{askToolName}}`; MUST use `xd://propose` write. MUST continue until decision-complete. +NEVER request plan approval via prose/{{#if askAvailable}}`{{askToolName}}`{{else}}a question{{/if}}; MUST use `xd://propose` write. MUST continue until decision-complete. diff --git a/packages/coding-agent/src/prompts/system/project-prompt.md b/packages/coding-agent/src/prompts/system/project-prompt.md index e8b7b91f1..a35e4a0c0 100644 --- a/packages/coding-agent/src/prompts/system/project-prompt.md +++ b/packages/coding-agent/src/prompts/system/project-prompt.md @@ -34,14 +34,14 @@ Context files above auto-loaded. NEVER `grep`/`glob` for `AGENTS.md`, `CLAUDE.md Working-directory layout: newest mtime first; depth ≤ 3. {{workspaceTree.rendered}} {{#if workspaceTree.truncated}} -Some entries elided to shorten tree — use `glob`/`read` to drill in. +{{#has tools "glob"}}{{#has tools "read"}}Some entries elided to shorten tree — use `{{toolRefs.glob}}`/`{{toolRefs.read}}` to drill in.{{/has}}{{/has}} {{/if}} {{/if}} {{/if}} {{#if additionalWorkspaceRoots.length}} -Additional workspace directories. This CURRENT workspace state supersedes workspace changes mentioned earlier in the conversation. Use absolute paths under these roots to `read`/`grep`/`glob`/`edit`. Manage with `/add-dir` and `/remove-dir`; `/dirs` lists them. +Additional workspace directories. This CURRENT workspace state supersedes workspace changes mentioned earlier in the conversation. {{#ifAny (includes tools "read") (includes tools "grep") (includes tools "glob") (includes tools "edit")}}Use absolute paths under these roots to {{#has tools "read"}}`{{toolRefs.read}}`{{/has}}{{#has tools "grep"}}{{#ifAny (includes tools "read")}}/{{/ifAny}}`{{toolRefs.grep}}`{{/has}}{{#has tools "glob"}}{{#ifAny (includes tools "read") (includes tools "grep")}}/{{/ifAny}}`{{toolRefs.glob}}`{{/has}}{{#has tools "edit"}}{{#ifAny (includes tools "read") (includes tools "grep") (includes tools "glob")}}/{{/ifAny}}`{{toolRefs.edit}}`{{/has}}.{{/ifAny}} Manage with `/add-dir` and `/remove-dir`; `/dirs` lists them. {{#each additionalWorkspaceRoots}} - {{this}} {{/each}} diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 58fe60192..ad26d0f2b 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -124,9 +124,11 @@ MUST use specialized tool over shell equivalent: {{#has tools "bash"}}- Bash litmus: one external-CLI call/short pipeline returning count, frequency, set difference, checksum. For merely moving, paging, trimming fetchable bytes: tool.{{/has}} {{#if autoQaEnabled}} +{{#has tools "write"}} `{{toolRefs.write}} xd://report_issue`: automated QA. Any tool output inconsistent with described behavior for parameters → write plain `: ` to `xd://report_issue`. False positives fine. +{{/has}} {{/if}} # Exploration @@ -179,8 +181,9 @@ Delegation preferred. Once design settles, SHOULD fan substantial work to `{{too - Tool failure/file change since read → re-read before acting. # 3. Decompose -- Update todos; skip trivial requests. +{{#has tools "todo"}}- Update todos; skip trivial requests. - Todo calls NEVER alone: batch each with turn's real calls (`init` with first reads/edits; `done` with next action/final verification). Todo-only assistant turn wastes round trip. +{{/has}} # 4. Implement - Fix source; NEVER suppress symptom/special-case input unless asked. @@ -191,7 +194,17 @@ Delegation preferred. Once design settles, SHOULD fan substantial work to `{{too # 5. Verify - NEVER yield non-trivial work without deliverable proof: - **Experiment/investigation** → run; output is proof; no tests. - - **UI change** → browser-drive; visual confirmation is proof; no tests unless existing suite really breaks. + - **UI change** → verify against the actual surface: +{{#has tools "browser"}} + - **Web UI** → browser-drive with `{{toolRefs.browser}}`; visual confirmation is proof; no tests unless existing suite really breaks. +{{/has}} +{{#has tools "computer"}} + - **Native desktop UI** → drive with `{{toolRefs.computer}}`; ground every claim in fresh screenshot or accessibility evidence. +{{/has}} + - **TUI/CLI** → launch the actual program and verify terminal interaction, output, or state. +{{#ifAny (not (includes tools "browser")) (not (includes tools "computer"))}} + - No suitable runtime tool for the changed surface → verify with a behavioral test or smoke test; explicitly report when visual verification cannot be performed. +{{/ifAny}} - **Bug fix** → reproduce, fix, confirm reproduction no longer triggers. - **Permanent feature/API change** → existing changed-contract tests. Add test only for uncovered new observable contract or user request. - Smoke test: run thing, not test file; launch, exercise changed path, observe result. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 4756dd5c6..61f851008 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -148,7 +148,7 @@ import type { IrcMessage } from "../irc/bus"; import type { DaemonCompletionNotification } from "../launch/protocol"; import { shutdownMnemopiEmbedClient } from "../mnemopi/embed-client"; import { getMnemopiSessionState, type MnemopiSessionState, setMnemopiSessionState } from "../mnemopi/state"; -import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate"; +import { containsOrchestrate, renderOrchestrateNotice } from "../modes/orchestrate"; import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; @@ -4955,12 +4955,15 @@ export class AgentSession { : sessionPlanUrl; const planExists = fs.existsSync(resolvedPlanPath); + const activeToolNames = this.getActiveToolNames(); const content = prompt.render(planModeActivePrompt, { planFilePath: displayPlanPath, planExists, askToolName: "ask", writeToolName: "write", editToolName: "edit", + askAvailable: activeToolNames.includes("ask"), + taskAvailable: activeToolNames.includes("task"), isHashlineEditMode: this.#resolveActiveEditMode() === "hashline", reentry: state.reentry ?? false, iterative: state.workflow === "iterative", @@ -5081,14 +5084,19 @@ export class AgentSession { }); } if (this.#magicKeywordEnabled("orchestrate") && containsOrchestrate(text)) { - keywordNotices.push({ - role: "custom", - customType: "orchestrate-notice", - content: ORCHESTRATE_NOTICE, - display: false, - attribution: "user", - timestamp, - }); + const activeToolNames = this.getActiveToolNames(); + // The contract is entirely about `task` subagent dispatch; without the + // task tool the notice would demand an unavailable capability. + if (activeToolNames.includes("task")) { + keywordNotices.push({ + role: "custom", + customType: "orchestrate-notice", + content: renderOrchestrateNotice({ tools: activeToolNames }), + display: false, + attribution: "user", + timestamp, + }); + } } if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) { const activeToolNames = this.getActiveToolNames(); diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts index 86e62affc..e38fc8fcb 100644 --- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts @@ -171,6 +171,18 @@ describe("AgentSession magic keyword settings", () => { expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]); }); + it("skips orchestrate notice when the task tool is inactive", async () => { + const created = await createMagicKeywordSession(root, []); + session = created.session; + authStorage = created.authStorage; + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + + await session.prompt("please orchestrate this"); + + const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ customType?: string }>; + expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]); + }); + it("skips workflowz notice when the eval tool is inactive", async () => { const created = await createMagicKeywordSession(root, [mockTaskTool]); session = created.session; diff --git a/packages/coding-agent/test/modes/orchestrate.test.ts b/packages/coding-agent/test/modes/orchestrate.test.ts index c1cac442a..e5ad21e27 100644 --- a/packages/coding-agent/test/modes/orchestrate.test.ts +++ b/packages/coding-agent/test/modes/orchestrate.test.ts @@ -2,7 +2,7 @@ import { beforeAll, describe, expect, it } from "bun:test"; import { containsOrchestrate, highlightOrchestrate, - ORCHESTRATE_NOTICE, + renderOrchestrateNotice, } from "@oh-my-pi/pi-coding-agent/modes/orchestrate"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { containsUltrathink, highlightUltrathink } from "@oh-my-pi/pi-coding-agent/modes/ultrathink"; @@ -85,11 +85,37 @@ describe("orchestrate keyword highlighting", () => { describe("orchestrate notice", () => { it("is a self-contained system notice carrying the orchestration contract", () => { - expect(ORCHESTRATE_NOTICE.startsWith("")).toBe(true); - expect(ORCHESTRATE_NOTICE.endsWith("")).toBe(true); - expect(ORCHESTRATE_NOTICE).toContain("orchestrator"); + const notice = renderOrchestrateNotice({ + tools: ["read", "task", "edit", "write", "lsp", "bash", "todo"], + }); + expect(notice.startsWith("")).toBe(true); + expect(notice.endsWith("")).toBe(true); + expect(notice).toContain("orchestrator"); // The contract must not retain the slash-command input placeholder. - expect(ORCHESTRATE_NOTICE).not.toContain("$@"); + expect(notice).not.toContain("$@"); + }); + + it("omits tool-budget mentions for tools absent from the session", () => { + const notice = renderOrchestrateNotice({ tools: ["read"] }); + expect(notice).not.toContain("`task` for dispatch"); + expect(notice).not.toContain("`edit`"); + expect(notice).not.toContain("`write`"); + expect(notice).not.toContain("`lsp diagnostics`"); + expect(notice).not.toContain("via `bash`"); + expect(notice).not.toContain("`todo` for tracking"); + }); + + it("does not name edit when only write is available", () => { + const writeOnly = renderOrchestrateNotice({ tools: ["read", "write"] }); + expect(writeOnly).toContain("with `write`"); + expect(writeOnly).not.toContain("`edit`/`write`"); + expect(writeOnly).not.toContain("with `edit`"); + }); + + it("does not name write when only edit is available", () => { + const editOnly = renderOrchestrateNotice({ tools: ["read", "edit"] }); + expect(editOnly).toContain("with `edit`"); + expect(editOnly).not.toContain("`edit`/`write`"); }); }); diff --git a/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts b/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts index 00cfffb4e..f2313d0ab 100644 --- a/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts +++ b/packages/coding-agent/test/plan-mode/reentry-prompt.test.ts @@ -9,15 +9,57 @@ const BASE = { editToolName: "edit", isHashlineEditMode: false, iterative: false, + askAvailable: true, + taskAvailable: true, + scoutAvailable: true, + reentry: false, + planExists: true, } as const; -function render(overrides: { reentry: boolean; planExists: boolean }): string { +type Overrides = Partial>; + +function render(overrides: Overrides = {}): string { return prompt.render(planModeActivePrompt, { ...BASE, ...overrides }); } describe("plan-mode re-entry prompt", () => { it("only emits the Re-entry section when re-entering", () => { - expect(render({ reentry: false, planExists: true })).not.toContain("## Re-entry"); - expect(render({ reentry: true, planExists: true })).toContain("## Re-entry"); + expect(render({ reentry: false })).not.toContain("## Re-entry"); + expect(render({ reentry: true })).toContain("## Re-entry"); + }); +}); + +describe("plan-mode-active tool availability", () => { + it("omits ask-tool directives when ask is unavailable", () => { + const withoutAsk = render({ askAvailable: false, iterative: true }); + expect(withoutAsk).not.toContain("`ask` with 2–4 mutually exclusive options"); + expect(withoutAsk).not.toContain("use `ask` for preferences and tradeoffs"); + expect(withoutAsk).not.toContain("Using `ask` to gather requirements"); + + const withAsk = render({ askAvailable: true, iterative: true }); + expect(withAsk).toContain("`ask` with 2–4 mutually exclusive options"); + expect(withAsk).toContain("use `ask` for preferences and tradeoffs"); + }); + + it("records preferences as assumptions when ask is unavailable", () => { + const iterativeWithoutAsk = render({ askAvailable: false, iterative: true }); + expect(iterativeWithoutAsk).toContain("Record them as Assumptions with a recommended default"); + expect(iterativeWithoutAsk).toContain("record preferences and tradeoffs as Assumptions"); + expect(iterativeWithoutAsk).not.toContain("`ask` for preferences and tradeoffs only"); + + const parallelWithoutAsk = render({ askAvailable: false, iterative: false }); + expect(parallelWithoutAsk).toContain( + "record any remaining preference questions as Assumptions with a recommended default", + ); + // A prose question cannot end the turn in plan mode — no prose-terminal option. + expect(parallelWithoutAsk).not.toContain("Presenting a choice between approaches"); + }); + + it("omits scout-via-task dispatch when the task tool is unavailable", () => { + const withoutTask = render({ taskAvailable: false, scoutAvailable: true }); + expect(withoutTask).not.toContain("(via `task`)"); + + const withTask = render({ taskAvailable: true, scoutAvailable: true }); + expect(withTask).toContain("(via `task`)"); }); }); diff --git a/packages/coding-agent/test/system-prompt-inventory.test.ts b/packages/coding-agent/test/system-prompt-inventory.test.ts index 67933b61a..ba2ededd4 100644 --- a/packages/coding-agent/test/system-prompt-inventory.test.ts +++ b/packages/coding-agent/test/system-prompt-inventory.test.ts @@ -682,4 +682,70 @@ describe("system prompt tool inventory", () => { expect(withScout).toContain("one read-only scout while working is allowed"); expect(withoutScout).not.toContain("read-only scout"); }); + + it("does not require browser verification when the browser tool is absent (issue #8139)", async () => { + const opts = { + cwd: tempDir, + contextFiles: [], + skills: [], + rules: [], + workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, + }; + const tools = new Map(TOOLS); + const withoutBrowser = ( + await buildSystemPrompt({ + ...opts, + toolNames: ["read", "bash"], + tools, + nativeTools: true, + inlineToolDescriptors: false, + }) + ).systemPrompt.join("\n\n"); + + expect(withoutBrowser).not.toContain("drive it in `browser`"); + expect(withoutBrowser).not.toContain("drive it in browser"); + expect(withoutBrowser).toContain("TUI/CLI"); + expect(withoutBrowser).toContain("behavioral test or smoke test"); + + tools.set("browser", { + label: "Browser", + description: "Drives a real Chromium tab.", + parameters: { type: "object", properties: {} }, + }); + const withBrowser = ( + await buildSystemPrompt({ + ...opts, + toolNames: ["read", "bash", "browser"], + tools, + nativeTools: true, + inlineToolDescriptors: false, + }) + ).systemPrompt.join("\n\n"); + + expect(withBrowser).toContain("drive it in `browser`"); + // A browser-only session still needs the smoke-test fallback for + // native-desktop surfaces (no computer tool). + expect(withBrowser).toContain("behavioral test or smoke test"); + }); + + it("omits todo workflow guidance when the todo tool is absent", async () => { + const opts = { + cwd: tempDir, + contextFiles: [], + skills: [], + rules: [], + workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, + tools: TOOLS, + nativeTools: true, + inlineToolDescriptors: false, + }; + const withoutTodo = (await buildSystemPrompt({ ...opts, toolNames: ["read", "bash"] })).systemPrompt.join("\n\n"); + expect(withoutTodo).not.toContain("Todo calls NEVER travel alone"); + expect(withoutTodo).not.toContain("batch every todo op"); + + const withTodo = (await buildSystemPrompt({ ...opts, toolNames: ["read", "bash", "todo"] })).systemPrompt.join( + "\n\n", + ); + expect(withTodo).toContain("Todo calls NEVER travel alone"); + }); });