diff --git a/docs/tools/task.md b/docs/tools/task.md index a03e492b1..383d9c595 100644 --- a/docs/tools/task.md +++ b/docs/tools/task.md @@ -1,6 +1,6 @@ # task -> Launch subagents for parallel, optionally isolated work. +> Spawn one subagent per call to work in the background, or resume an existing one. ## Source - Entry: `packages/coding-agent/src/task/index.ts` @@ -9,217 +9,163 @@ - `packages/coding-agent/src/task/types.ts` — dynamic schema, progress/result types, output caps. - `packages/coding-agent/src/task/discovery.ts` — discover project/user/plugin/bundled agents. - `packages/coding-agent/src/task/agents.ts` — bundled agent definitions and frontmatter parsing. - - `packages/coding-agent/src/task/executor.ts` — create child sessions, run subagents, collect output. - - `packages/coding-agent/src/task/parallel.ts` — concurrency-limited scheduling and async semaphore. + - `packages/coding-agent/src/task/executor.ts` — create child sessions, run/resume subagents, collect output, hand finished sessions to the lifecycle manager. + - `packages/coding-agent/src/registry/agent-lifecycle.ts` — idle-TTL parking and revival of finished subagents. + - `packages/coding-agent/src/registry/agent-registry.ts` — process-global agent directory (`running | idle | parked | aborted`). + - `packages/coding-agent/src/async/job-manager.ts` — background job registration, progress, and result delivery. + - `packages/coding-agent/src/task/parallel.ts` — `Semaphore` used for the session-scoped concurrency bound. - `packages/coding-agent/src/task/isolation-backend.ts` — isolation backend resolution and platform fallback. - `packages/coding-agent/src/task/worktree.ts` — worktree / FUSE / ProjFS setup, patch capture, branch merge. - `packages/coding-agent/src/task/output-manager.ts` — session-scoped `agent://` id allocation. - - `packages/coding-agent/src/task/simple-mode.ts` — `default` / `schema-free` / `independent` field gating. + - `packages/coding-agent/src/task/name-generator.ts` — default AdjectiveNoun agent ids. + - `packages/coding-agent/src/task/simple-mode.ts` — `default` / `schema-free` / `independent` schema gating. - `packages/coding-agent/src/internal-urls/agent-protocol.ts` — resolve `agent://` to saved subagent output. + - `packages/coding-agent/src/internal-urls/history-protocol.ts` — resolve `history://` to a concise transcript. - `packages/coding-agent/src/tools/index.ts` — tool registration and recursion-depth gating. - `packages/coding-agent/src/sdk.ts` — child-session router/tool wiring and per-subagent `AgentOutputManager`. - `docs/task-agent-discovery.md` — deeper discovery and precedence notes. - - `docs/handoff-generation-pipeline.md` — session artifact/handoff persistence patterns used by the wider session layer. ## Inputs -### Default mode (`task.simple = "default"`) +One call spawns (or resumes) exactly one subagent. There is no batch parameter and no shared `context` parameter — shared background goes into a `local://` file (e.g. `local://ctx.md`) that each assignment references; subagents share the parent's `local://` root. | Field | Type | Required | Description | | --- | --- | --- | --- | -| `agent` | `string` | Yes | Exact agent name for every task item. Resolved at execution time through `discoverAgents(...)`. | -| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Batch of small, self-contained task items. `id` max length 48 in schema; duplicate ids are rejected case-insensitively at runtime. | -| `context` | `string` | No | Shared background prepended to every subagent system prompt. Trimmed before use. | -| `schema` | `string` | No | JSON-encoded JTD schema. Overrides agent/session output schema when this mode allows task-level schemas. | -| `isolated` | `boolean` | No | Only present when the tool is created with isolation enabled. Requests isolated execution for the whole batch. | +| `agent` | `string` | Conditional | Agent type to spawn. Required unless `resume` is set; providing both is a validation error. | +| `resume` | `string` | Conditional | Existing agent id — revive the agent if parked and run a follow-up assignment in its existing session. Cannot be combined with `agent` or `isolated`. | +| `id` | `string` | No | Stable agent id, schema max length 48. Defaults to a generated AdjectiveNoun name. Uniquified per session by `AgentOutputManager`. | +| `description` | `string` | No | UI label only; the subagent never sees it. | +| `assignment` | `string` | Yes | The work — complete, self-contained instructions. Empty-after-trim is rejected. | +| `schema` | `string` | No | JSON-encoded JTD schema for the expected `yield` payload. Field exists only when `task.simple = "default"`. | +| `isolated` | `boolean` | No | Run in an isolated workspace and return patches. Field exists only when `task.isolation.mode` is not `none`. Isolated agents are NOT resumable. | -`tasks[].description` is UI-only. `tasks[].assignment` is the actual per-task instruction. - -### Schema-free mode (`task.simple = "schema-free"`) - -Same as default, except `schema` is rejected by `validateTaskModeParams(...)` in `packages/coding-agent/src/task/index.ts`. - -### Independent mode (`task.simple = "independent"`) - -| Field | Type | Required | Description | -| --- | --- | --- | --- | -| `agent` | `string` | Yes | Exact agent name. | -| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Same item shape, but each `assignment` must carry all required background because shared `context` is disabled. | -| `isolated` | `boolean` | No | Same conditional field as above. | - -In this mode both `context` and `schema` are rejected. +Simple-mode gating (`task.simple`, one axis): `default` accepts the per-call `schema` override; `schema-free` and `independent` reject it (`validateTaskModeParams(...)`). `independent` additionally renders the subagent user prompt with the independent-mode flag. Agent frontmatter and inherited session schemas work in every mode. ## Outputs + The tool returns one text block plus `details: TaskToolDetails`. -`details` fields: -- `projectAgentsDir: string | null` — nearest discovered project `agents/` dir. -- `results: SingleResult[]` — one entry per task in input order for synchronous execution; empty for async-launch responses. -- `totalDurationMs: number` -- `usage?: Usage` — sum of per-subagent assistant-message usage. -- `outputPaths?: string[]` — written `.md` artifact paths for completed subagent outputs. -- `progress?: AgentProgress[]` — live or final per-task progress snapshots. -- `async?: { state: "running" | "completed" | "failed"; jobId: string; type: "task" }` — present for background execution updates/results. +Immediate (async) response — the normal case: +- `content`: `` Spawned agent `` (job ``). The result will be delivered when it yields. ... `` (or `Resumed agent ...`), plus a coordination hint (`irc` DM when enabled, otherwise `job`). +- `details`: `{ projectAgentsDir: null, results: [], totalDurationMs: 0, progress: [], async: { state: "running", jobId, type: "task" } }`. +- Live progress keeps streaming into the same tool block via `onUpdate(...)`; the final result arrives later as an async-result injection into the parent conversation. The delivery text appends a resume hint: `` is now idle — task(resume:"") to continue it, transcript at history:// `` (aborted variant points at the transcript only). + +Settled (sync-fallback or job-body) response: +- `content`: summary rendered from `packages/coding-agent/src/prompts/tools/task-summary.md` with a preview capped at 5000 chars; `agent://` holds the full output. +- `details.results`: at most one `SingleResult`; `usage`, `outputPaths` populated. `SingleResult` includes: - identity: `index`, `id`, `agent`, `agentSource`, `description`, optional `assignment` -- status: `exitCode`, optional `error`, optional `aborted`, optional `abortReason` -- output: `output`, `stderr`, `truncated`, `durationMs`, `tokens` +- status: `exitCode`, optional `error`, optional `aborted`, optional `abortReason`, optional `retryFailure` +- output: `output`, `stderr`, `truncated`, `durationMs`, `tokens`, `requests`, optional `contextTokens`/`contextWindow` - artifact metadata: `outputPath?`, `patchPath?`, `branchName?`, `nestedPatches?`, `outputMeta?` - extracted tool data: `extractedToolData?` from registered subprocess tool handlers such as `yield` and `report_finding` Artifacts and side channels: -- Every subagent with an artifacts dir writes `.md`; `agent://` resolves to that file. -- If the output file is JSON, `agent:///` and `agent://?q=` perform JSON extraction in `packages/coding-agent/src/internal-urls/agent-protocol.ts`. -- When the parent session persists artifacts, each subagent also gets `.jsonl` session history. -- Isolated patch mode writes `.patch` per successful task before merge. -- Async mode returns immediately after job registration, then emits `onUpdate(...)` progress snapshots and later hands completion to the session async-job pipeline. +- Every subagent with an artifacts dir writes `.md`; `agent://` resolves to that file. Resumes overwrite it per assignment. +- If the output file is JSON, `agent:///` and `agent://?q=` perform JSON extraction. +- Each subagent gets `.jsonl` session history when the parent persists artifacts; `history://` renders it as a concise transcript (works for live and parked agents). +- Isolated patch mode writes `.patch` before merge. ## Flow -1. `TaskTool.create(...)` in `packages/coding-agent/src/task/index.ts` calls `discoverAgents(session.cwd)` once to build the dynamic prompt description from current agents and `task.simple` capabilities. -2. `execute(...)` validates mode-gated fields with `validateTaskModeParams(...)`. -3. It decides async vs sync: - - sync when `async.enabled` is false - - sync when the selected cached agent has `blocking === true` - - sync when `tasks.length === 0` - - otherwise async job scheduling -4. Async path: - - allocate unique output ids with `AgentOutputManager.allocateBatch(...)` - - create one async job per task through `session.asyncJobManager.register(...)` - - limit concurrent job bodies with `Semaphore(task.maxConcurrency)` from `packages/coding-agent/src/task/parallel.ts` - - each job body calls `#executeSync(...)` with a one-task batch and the preallocated id - - `onUpdate(...)` emits aggregate `progress` snapshots and `details.async` -5. Sync path (`#executeSync(...)`) rediscovers agents from disk via `discoverAgents(...)`, so runtime resolution can differ from the earlier prompt description. -6. It resolves the requested agent with `getAgent(...)`, rejects unknown or disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention. -7. It derives the effective output schema in priority order: task call `schema` (if allowed) → agent frontmatter `output` → inherited parent session schema. -8. It validates task ids: missing ids and case-insensitive duplicates are immediate errors. -9. If `isolated` was requested, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`) and resolves the actual backend through `resolveIsolationBackendForTaskExecution(...)`. -10. It chooses an artifacts dir from the parent session when available, otherwise a temp dir, and writes `context.md` there when `session.getCompactContext?.()` returns content. -11. It allocates unique ids again if the caller did not preallocate them, then builds `tasksWithUniqueIds`. -12. For each task, it seeds an `AgentProgress` entry and runs `runTask(...)` through `mapWithConcurrencyLimit(...)` using `task.maxConcurrency`. -13. Non-isolated `runTask(...)` calls `runSubprocess(...)` directly with parent cwd. -14. Isolated `runTask(...)`: - - creates an isolation workspace (`ensureWorktree(...)`, `ensureFuseOverlay(...)`, or `ensureProjfsOverlay(...)`) - - applies the captured baseline for worktrees - - runs `runSubprocess(...)` inside that workspace - - on success, either commits to a per-task branch (`mergeMode === "branch"`) or captures a patch with `captureDeltaPatch(...)` - - always cleans up the isolation workspace/backend -15. `runSubprocess(...)` in `packages/coding-agent/src/task/executor.ts` creates a child agent session with: - - isolated settings snapshot via `Settings.isolated(...)`, forcing `async.enabled = false` and `bash.autoBackground.enabled = false` - - child `agentId` / `parentTaskPrefix` equal to the allocated task id - - child internal URL router and `AgentOutputManager` from `packages/coding-agent/src/sdk.ts` - - the shared `context`, optional `context.md` reference, optional isolation worktree path, output schema, and IRC peer roster in the system prompt template -16. Child tool availability is derived from the agent definition plus runtime guards: - - explicit `agent.tools` if provided - - auto-add `task` when the agent has `spawns` and recursion depth allows it - - remove `task` at or past `task.maxRecursionDepth` - - expand `exec` to `eval` and `bash` - - strip parent-owned `todo` after session creation -17. `runSubprocess(...)` subscribes to child agent events, coalesces progress updates every 150 ms, forwards lifecycle/progress events on the parent event bus, and extracts tool data through `subprocessToolRegistry`. -18. The child must finish through the hidden `yield` tool. If it does not, `runSubprocess(...)` sends up to 3 reminder prompts; the last reminder forces `toolChoice = yield` when supported. -19. Finalization uses `finalizeSubprocessOutput(...)` to reconcile raw assistant text, `yield` payloads, structured schemas, `report_finding` data, and abort states. Output is truncated with `MAX_OUTPUT_BYTES` / `MAX_OUTPUT_LINES` before returning to the parent, but the full raw output is still written to `.md`. -20. After all sync tasks finish, `#executeSync(...)` aggregates usage, collects artifact paths, and if isolation was used merges results back: - - branch mode: cherry-pick per-task branches with `mergeTaskBranches(...)`, then delete merged branches with `cleanupTaskBranches(...)` - - patch mode: combine non-empty patch artifacts, dry-check with `git.patch.canApplyText(...)`, then apply or leave manual artifacts - - nested repo patches are applied separately with `applyNestedPatches(...)` -21. The final text summary is rendered from `packages/coding-agent/src/prompts/tools/task-summary.md` and includes `agent://` handles for outputs that exist. +1. `TaskTool.create(...)` discovers agents once per cwd through a process-level memo (`discoverAgentsForCreate`) to render the dynamic prompt description. +2. `execute(...)` repairs raw params (`repairTaskParams`), then validates: schema gating per `task.simple`, `agent` XOR `resume`, `resume` excludes `isolated`, non-empty `assignment`. +3. Sync fallback only when the session has no `AsyncJobManager` (orphaned host) or the selected agent definition declares `blocking: true`; the call then runs `#executeSync(...)` inline under the session-scoped semaphore. +4. Otherwise execution is always async: + - the agent id is resolved up front — `resume` must name a registered agent (else `ToolError` pointing at `irc` op:"list" and `history://`); spawns allocate via `AgentOutputManager.allocate(params.id || generateTaskName())`; + - one `type: "task"` job is registered with `session.asyncJobManager` (`id` = agent id, `queued: true`, `ownerId` = caller agent id) and the tool returns immediately; + - the job body acquires the session-scoped `Semaphore` (one per `TaskTool` instance, sized from `task.maxConcurrency` at first use), marks the job running, runs `#executeSync(...)`, and reports progress through `buildAsyncDetails`/`onUpdate`; + - a failed or aborted run throws `TaskJobError` so the job lands `failed`, but the agent itself stays registered and interrogable. +5. `#executeSync(...)` dispatches: `resume` → `#executeResume(...)`, else `#runSpawn(...)`. +6. Resume path (`#executeResume`): + - `AgentLifecycleManager.global().ensureLive(resumeId)` returns the live session, reviving a parked one from its session JSONL; unknown ids or parked-without-reviver throw a `ToolError`; + - `resumeSubprocess(...)` in `packages/coding-agent/src/task/executor.ts` injects the rendered follow-up through the session's normal prompt path and drives it through the same monitor/yield/finalize pipeline as a spawn; + - the session is never disposed here — registry status settles back to `idle` (even on failure/abort) and the lifecycle manager re-arms the idle TTL. +7. Spawn path (`#runSpawn`) rediscovers agents from disk, so runtime resolution can differ from the create-time description. +8. It resolves the requested agent, rejects unknown or settings-disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention. +9. Output schema priority: task call `schema` (when `task.simple` allows) → agent frontmatter `output` → inherited parent session schema. +10. Plan mode swaps in an `effectiveAgent` with a read-only tool subset and plan-mode prompt; `runSubprocess(...)` receives the effective agent. +11. If `isolated`, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`) and resolves the backend through isolation-backend resolution with platform fallback. +12. Artifacts dir comes from the parent session file when available, otherwise a temp dir. When the session is executing an approved plan, the plan reference is handed to the subagent. +13. Non-isolated spawns call `runSubprocess(...)` directly with parent cwd; isolated spawns run inside the isolation workspace, then commit to a branch (`mergeMode === "branch"`) or capture a patch, and always clean up the workspace. +14. `runSubprocess(...)` creates a child agent session with an isolated settings snapshot (forcing `async.enabled = false` and `bash.autoBackground.enabled = false` — subagents are internally synchronous), child `agentId` equal to the allocated id, child internal URL router/`AgentOutputManager`, output schema, and the IRC peer roster in the system prompt. +15. Child tool availability: explicit `agent.tools` if provided; auto-add `task` when the agent has `spawns` and depth allows; strip `task` at `task.maxRecursionDepth`; expand `exec` to `eval` + `bash`; strip parent-owned `todo`. +16. The child must finish through the hidden `yield` tool; up to 3 reminder prompts, the last forcing `toolChoice = yield` when supported. `finalizeSubprocessOutput(...)` reconciles raw text, `yield` payloads, structured schemas, `report_finding` data, and abort states. +17. End-of-run lifecycle (keep-alive, in `runSubprocess`'s finalizer): + - hard abort (caller signal / wall-clock / budget) → registry status `aborted`, session disposed — terminal; + - isolated run → status `parked` without a reviver (workspace is merged + cleaned, so the session is not resumable; transcript stays readable via `history://`), then session disposed and detached; + - everything else (success and failure alike) → status `idle` with the live session attached, and `AgentLifecycleManager.global().adopt(id, { idleTtlMs, revive })` arms the park timer. The reviver reopens the session JSONL (park closed the writer, so the single-writer lock is taken cleanly). +18. Lifecycle thereafter: `idle` agents are parked after `task.agentIdleTtlMs` (session disposed; `AgentRef` + session file retained); messaging (`irc`), `task(resume:)`, or the Agent Hub revives them back to `idle`. `"Main"` is never parked. ## Modes / Variants - Execution mode - - Sync inline execution — default path. - - Async background execution — one async job per task item when `async.enabled` is on and the chosen agent is not marked `blocking`. -- Simple mode - - `default` — accepts shared `context` and per-call `schema`. - - `schema-free` — accepts `context`, rejects `schema`. - - `independent` — rejects `context` and `schema`; each assignment stands alone. -- Isolation backend - - `none` — no isolation. - - `worktree` — detached git worktree plus baseline replay. - - `fuse-overlay` — Unix FUSE overlay mount. - - `fuse-projfs` — Windows ProjFS overlay. -- Isolation merge strategy - - Patch mode — capture/apply root patches, keep patch artifacts when application fails. - - Branch mode — commit each task onto `omp/task/` branch, cherry-pick into parent, preserve failed branches for manual resolution. -- Agent source - - Project custom agents — nearest project config/plugin agent directories, first by source-family precedence. - - User custom agents — user config/plugin agent directories after project dirs of the same source family. - - Bundled agents — appended last from `packages/coding-agent/src/task/agents.ts`. -- Bundled agent types - - `explore` — read-only scout with structured handoff output. - - `plan` — architecture/planning agent; may spawn `explore`. - - `designer` — UI/UX specialist. - - `reviewer` — review agent with `report_finding` extraction. - - `task` — general-purpose worker with full capabilities. - - `quick_task` — low-reasoning mechanical worker using the same task prompt body. - - `librarian` — source-grounded external API/library researcher. - - `oracle` — senior-engineer implementation/debugging/general consultation agent. + - Always-async background job — default; spawn and resume both go through `AsyncJobManager`. + - Sync inline fallback — only when no job manager exists or the agent definition has `blocking: true`. +- Spawn vs resume + - `agent: ""` — fresh subagent with a new (or caller-provided) id. + - `resume: ""` — follow-up assignment in an existing session; revives a parked agent first. Transcript accretes; `agent://` is overwritten per assignment. +- Simple mode (`task.simple`) + - `default` — accepts per-call `schema`. + - `schema-free` / `independent` — reject `schema`; `independent` also flags the subagent user prompt as independent-mode. +- Isolation backend: `none`, `worktree`, `fuse-overlay`, `fuse-projfs`. +- Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/`, cherry-pick into parent). +- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `quick_task`, `librarian`, `oracle`). ## Side Effects - Filesystem - - Writes `context.md`, `.jsonl`, and `.md` under the session artifacts dir or a temp task dir. - - In isolated patch mode writes `.patch` artifacts. - - Creates/removes worktrees or overlay mount directories. - - In branch mode creates temporary worktrees and task branches. + - Writes `.jsonl` and `.md` under the session artifacts dir or a temp task dir; isolated patch mode writes `.patch`. + - Creates/removes worktrees or overlay mount directories; branch mode creates temporary worktrees and task branches. - Network - Child sessions may use whichever networked tools/models their active tool set permits. - MCP proxy tools can call existing parent MCP connections with a 60_000 ms timeout. - Subprocesses / native bindings - - `fuse-overlayfs` and `fusermount`/`fusermount3` for FUSE isolation. - - ProjFS native bindings via `@oh-my-pi/pi-natives` on Windows. + - `fuse-overlayfs` and `fusermount`/`fusermount3` for FUSE isolation; ProjFS native bindings on Windows. - Git operations for baseline capture, patch apply, worktrees, branches, stash, cherry-pick, commits. - Session state (transcript, memory, jobs, checkpoints, registries) - - Creates child `AgentSession` instances with isolated settings snapshots. - - Registers async jobs in `session.asyncJobManager` for background task mode. + - Creates child `AgentSession` instances with isolated settings snapshots; finished sessions stay registered in the process-global `AgentRegistry` as `idle`/`parked` until process teardown or explicit release. + - Registers one async job per call in `session.asyncJobManager`; completion is injected into the parent as an async-result message. + - Arms idle-TTL timers in `AgentLifecycleManager` (unref'd; they never hold the process open). - Emits `task:subagent:event`, `task:subagent:progress`, and `task:subagent:lifecycle` on the parent event bus. - - Allocates session-scoped output ids through `AgentOutputManager` so `agent://` remains unique across invocations and resumes. - - Shares the parent `local://` root with subagents by passing `localProtocolOptions` through `createAgentSession(...)`. -- User-visible prompts / interactive UI - - Async mode streams aggregate progress updates. - - Missing-`yield` recovery sends up to three internal reminder prompts to the child session. - - Final summaries include `` blocks for isolation fallbacks or merge failures. + - Allocates session-scoped output ids through `AgentOutputManager` so `agent://` stays unique across invocations and resumes. + - Shares the parent `local://` root and `ArtifactManager` with subagents. - Background work / cancellation - - Parent abort stops scheduling new work, aborts active child sessions, and marks unscheduled tasks as skipped. - - Async jobs keep their own cancellation via `AsyncJobManager`. + - `job cancel` (or parent tool-call abort) cancels the job; a hard-aborted run lands `aborted` and is torn down. + - Missing-`yield` recovery sends up to three internal reminder prompts to the child session. ## Limits & Caps -- Per-subagent output truncation: `MAX_OUTPUT_BYTES = 500_000` and `MAX_OUTPUT_LINES = 5000` in `packages/coding-agent/src/task/types.ts`. Full raw output is still written to `.md` before truncation is returned to the caller. -- Progress coalescing in child execution: `PROGRESS_COALESCE_MS = 150` in `packages/coding-agent/src/task/executor.ts`. -- Recent output tail for progress: `RECENT_OUTPUT_TAIL_BYTES = 8 * 1024` and `recentOutput` keeps the last 8 non-empty lines in `packages/coding-agent/src/task/executor.ts`. -- Missing-`yield` reminder retries: `MAX_YIELD_RETRIES = 3` in `packages/coding-agent/src/task/executor.ts`. -- MCP proxy timeout: `MCP_CALL_TIMEOUT_MS = 60_000` in `packages/coding-agent/src/task/executor.ts`. -- Task id schema cap: `tasks[].id` `maxLength: 48` in `packages/coding-agent/src/task/types.ts`. -- Prompt text says ids should be `≤32` chars, but the runtime schema allows 48; this mismatch is real. -- Async/full sync parallelism both use `task.maxConcurrency` from settings: - - sync path: `mapWithConcurrencyLimit(...)` - - async path: `Semaphore(...)` around job bodies -- Recursion depth gate: `task.maxRecursionDepth` from settings; `packages/coding-agent/src/tools/index.ts` hides the `task` tool at or beyond the limit, and `runSubprocess(...)` also strips child `task` access at max depth. -- Final inline summary preview per task uses `fullOutputThreshold = 5000` chars in `packages/coding-agent/src/task/index.ts`; longer outputs are summarized while `agent://` points to the full artifact. +- Concurrency: one session-scoped `Semaphore` sized from `task.maxConcurrency` at first use (later setting changes do not resize it) bounds concurrent subagents across parallel `task` calls — both async job bodies and the sync fallback acquire it. +- Idle TTL: `task.agentIdleTtlMs`, default `420_000` ms (7 min); `<= 0` disables parking and keeps idle sessions live until exit. +- Per-subagent output truncation: `MAX_OUTPUT_BYTES = 500_000` and `MAX_OUTPUT_LINES = 5000` in `packages/coding-agent/src/task/types.ts` (overridable via `PI_TASK_MAX_OUTPUT_BYTES` / `PI_TASK_MAX_OUTPUT_LINES`). Full raw output is still written to `.md`. +- Progress coalescing: `PROGRESS_COALESCE_MS = 150`; recent-output tail: `RECENT_OUTPUT_TAIL_BYTES = 8 * 1024` (last 8 non-empty lines). +- Missing-`yield` reminder retries: `MAX_YIELD_RETRIES = 3`; MCP proxy timeout: `MCP_CALL_TIMEOUT_MS = 60_000` — both in `packages/coding-agent/src/task/executor.ts`. +- Agent id schema cap: `id` `maxLength: 48` in `packages/coding-agent/src/task/types.ts`. Prompt text says ids should be `≤32` chars; this mismatch is real. +- Soft request budget (`task.softRequestBudget`) and wall clock (`task.maxRuntimeMs`) apply to spawns and resumes alike. +- Recursion depth gate: `task.maxRecursionDepth`; `packages/coding-agent/src/tools/index.ts` hides the `task` tool at or beyond the limit, and `runSubprocess(...)` also strips child `task` access at max depth. +- Final inline summary preview uses `fullOutputThreshold = 5000` chars in `packages/coding-agent/src/task/index.ts`; `agent://` points to the full artifact. ## Errors -- Most validation failures are returned as normal tool text with empty `results`, not thrown: - - invalid simple-mode fields - - unknown/disabled agent - - missing tasks - - missing/duplicate task ids - - spawn-policy denial - - requesting `isolated` while isolation mode is `none` -- Isolated execution without a git repo returns `Isolated task execution requires a git repository. ...`. -- Backend resolution can return a hard error (`ProjFS isolation initialization failed...`) or a non-fatal warning with fallback to `worktree`. -- `mapWithConcurrencyLimit(...)` fails fast on non-abort worker exceptions; already completed results are preserved only in the thrown path’s local state, not surfaced unless the caller catches and converts them. -- Child-session failures surface as `SingleResult.exitCode = 1` with `stderr`/`error` populated. +- Parameter validation failures are returned as normal tool text with empty `results`: + - `schema` outside `task.simple = "default"` + - both or neither of `agent` / `resume` + - `resume` combined with `isolated` + - missing/empty `assignment` + - unknown or settings-disabled agent, spawn-policy denial, requesting `isolated` while isolation mode is `none` +- `resume` of an id not in the registry throws a `ToolError` naming `irc` op:"list" and `history://`. +- `ensureLive(...)` failures (agent parked without a reviver — e.g. an isolated run — or torn down) surface as `` Cannot resume "": ... `` `ToolError`s. +- Isolated execution without a git repo returns `Isolated task execution requires a git repository. ...`; backend resolution can hard-error (ProjFS init) or warn and fall back to `worktree`. +- Job registration failure returns `Failed to start background task job: ...`. +- Child failures surface as `SingleResult.exitCode = 1` with `stderr`/`error` populated; the async job is marked failed but the delivery text still carries the output plus a resume/transcript hint. - If the child omits `yield`, `finalizeSubprocessOutput(...)` injects warnings such as `SYSTEM WARNING: Subagent exited without calling yield tool after 3 reminders.` -- Async scheduling failures are accumulated per task; if no jobs start, the tool returns `Failed to start background task jobs: ...`. - `agent://` resolution errors are model-visible when another tool reads them: no session, no artifacts dir, missing id, conflicting extraction syntax, or invalid JSON for extraction. ## Notes -- Agent discovery precedence is first-wins by exact name: project dirs before user dirs within a source family, plugin agent dirs after config dirs, bundled agents last. See `packages/coding-agent/src/task/discovery.ts` and `docs/task-agent-discovery.md`. -- `TaskTool.create(...)` caches discovered agents only for description rendering and the async blocking-agent decision. `#executeSync(...)` rediscovers agents each call. -- Custom agent frontmatter can override bundled agents by name. Bundled definitions are embedded at build time in `packages/coding-agent/src/task/agents.ts`. -- Child sessions do not inherit conversation history automatically. The only built-in carry-over is shared `context`, optional `context.md`, workspace tree/skills/context files, and shared `local://` root. -- `Settings.isolated(...)` gives each child a session-isolated settings snapshot; tool enablement is recomputed inside the child session rather than sharing mutable parent tool state. -- When the parent passes `mcpManager`, child sessions disable standalone MCP discovery and instead get proxy tools that reuse the parent connections. -- Plan mode mutates an `effectiveAgent` with a read-only tool subset and plan-mode prompt text, but `runSubprocess(...)` is still invoked with `agent` rather than `effectiveAgent`. Model/thinking/schema overrides use the effective agent; prompt/tool/spawn restrictions do not fully flow through this call path. -- Branch-mode merge temporarily stashes the parent repo before cherry-picking task branches. A stash-pop conflict is treated as merge failure and leaves recovery state behind. -- Patch-mode only applies combined root patches if every successful task produced a patch and `git.patch.canApplyText(...)` succeeds. -- Nested git repos are handled separately from the root repo. They are copied into isolated worktrees, diffed independently, and merged later with `applyNestedPatches(...)` because parent git cannot track their file-level changes. -- `agent://` ids are name-based (`Task` first, `Task-2`/`Task-3` only when the name repeats, nested like `Parent.Child`) by `AgentOutputManager`; this is what prevents artifact collisions across repeated or nested task invocations. +- Parallelism is parallel `task` calls in one assistant message; the session-scoped semaphore bounds the fan-out. There is no batch array. +- Shared background convention: write it once to a `local://` file and reference that path in each assignment — subagents share the parent's `local://` root. This replaces the removed `context` parameter. +- Prefer `resume` over a fresh spawn for follow-up work: the resumed agent already holds the relevant context. `irc` op:"list" shows idle/parked candidates; `history://` shows what an agent has done. +- Subagents are internally synchronous: the executor forces `async.enabled = false` and `bash.autoBackground.enabled = false` in the child settings snapshot, so there are no fire-and-forget grandchildren. +- Agent discovery precedence is first-wins by exact name: project dirs before user dirs within a source family, plugin agent dirs after config dirs, bundled agents last. Create-time discovery is memoized per cwd for the prompt description; execution-time discovery stays fresh. +- Child sessions do not inherit conversation history. Built-in carry-over is the workspace tree/skills/context files, the shared `local://` root, and the approved-plan reference when one exists. +- When the parent passes `mcpManager`, child sessions disable standalone MCP discovery and get proxy tools that reuse parent connections. +- Branch-mode merge temporarily stashes the parent repo before cherry-picking; a stash-pop conflict is treated as merge failure and leaves recovery state behind. Patch mode only applies the combined root patch when `git.patch.canApplyText(...)` succeeds; failures leave the `.patch` artifact for manual handling. +- Nested git repos are diffed independently inside isolated workspaces and merged separately with `applyNestedPatches(...)`. +- `agent://` ids are name-based (`Task` first, `Task-2`/`Task-3` only when the name repeats, nested like `Parent.Child`) by `AgentOutputManager`; this is what prevents artifact collisions across repeated or nested invocations. diff --git a/packages/coding-agent/src/cli/gallery-cli.ts b/packages/coding-agent/src/cli/gallery-cli.ts index 7f4eb8c67..31d1a958b 100644 --- a/packages/coding-agent/src/cli/gallery-cli.ts +++ b/packages/coding-agent/src/cli/gallery-cli.ts @@ -69,7 +69,7 @@ function fakeToolFor(name: string, fixture: GalleryFixture | undefined): AgentTo if (!fixture?.label && !fixture?.editMode && !fixture?.customRendered) return undefined; const tool: Record = { name, label: fixture.label ?? name, mode: fixture.editMode }; if (fixture.customRendered) { - const renderer = toolRenderers[name] as + const renderer = toolRenderers[fixture.renderer ?? name] as | { renderCall?: unknown; renderResult?: unknown; mergeCallAndResult?: unknown; inline?: unknown } | undefined; if (renderer) { diff --git a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts index d1c262abe..df66493a9 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts @@ -1,62 +1,76 @@ -// Gallery fixtures for the agentic orchestration tools (task, goal, job). +// Gallery fixtures for the agentic orchestration tools (task, irc, goal, job). +import type { Usage } from "@oh-my-pi/pi-ai"; +import type { TaskToolDetails } from "../../task/types"; +import type { IrcDetails } from "../../tools/irc"; import type { GalleryFixture } from "./types"; +/** Message/activity timestamps are offsets from load time so gallery ages stay plausible. */ +const FIXTURE_NOW = Date.now(); + +/** Plausible cumulative usage for a fixture subagent run. */ +const fixtureUsage = (tokens: { input: number; output: number }, costTotal: number): Usage => ({ + input: tokens.input, + output: tokens.output, + cacheRead: 0, + cacheWrite: 0, + totalTokens: tokens.input + tokens.output, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: costTotal }, +}); + export const agenticFixtures: Record = { task: { label: "Task", customRendered: true, - // Streaming: agent chosen, first task fully arrived, second still landing. + // Streaming: agent chosen, assignment still landing. streamingArgs: { agent: "task", - tasks: [ - { - id: "AuthLoader", - description: "Load auth middleware", - assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie flow.", - }, - { id: "RateLimiter", description: "Audit rate limiter" }, - ], + id: "AuthLoader", + description: "Load auth middleware", + assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie", }, args: { agent: "task", - context: [ - "# Goal", - "Harden the HTTP auth stack before the release cut.", - "# Constraints", - "Touch only files under packages/server/src/auth/. Do not run gates.", - ].join("\n"), - tasks: [ - { - id: "AuthLoader", - description: "Load auth middleware", - assignment: - "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", - }, - { - id: "RateLimiter", - description: "Audit rate limiter", - assignment: - "Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.", - }, - { - id: "TokenRotation", - description: "Check token rotation", - assignment: - "Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.", - }, - ], + id: "AuthLoader", + description: "Load auth middleware", + assignment: + "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", }, result: { content: [ { type: "text", - text: "3 agents completed: AuthLoader, RateLimiter, TokenRotation.", + text: "Agent AuthLoader completed.", }, ], details: { projectAgentsDir: null, totalDurationMs: 48_200, - usage: { cost: { total: 0.34 } }, + usage: fixtureUsage({ input: 52_600, output: 8_800 }, 0.12), + progress: [ + { + index: 0, + id: "AuthLoader", + agent: "task", + agentSource: "bundled", + status: "completed", + task: "Read packages/server/src/auth/session.ts and middleware.ts", + description: "Load auth middleware", + lastIntent: "Documenting session-cookie flow", + recentTools: [ + { tool: "read", args: "packages/server/src/auth/session.ts", endMs: 1_749_200_040_000 }, + { tool: "read", args: "packages/server/src/auth/middleware.ts", endMs: 1_749_200_052_000 }, + ], + recentOutput: ["Session validation runs in middleware.ts:42 via verifySessionCookie()."], + toolCount: 9, + requests: 6, + tokens: 61_400, + contextTokens: 23_100, + contextWindow: 200_000, + cost: 0.12, + durationMs: 41_900, + resolvedModel: "anthropic/claude-sonnet", + }, + ], results: [ { index: 0, @@ -77,100 +91,31 @@ export const agenticFixtures: Record = { truncated: false, durationMs: 41_900, tokens: 61_400, + requests: 6, contextTokens: 23_100, contextWindow: 200_000, resolvedModel: "anthropic/claude-sonnet", - usage: { cost: { total: 0.12 } }, + usage: fixtureUsage({ input: 52_600, output: 8_800 }, 0.12), outputMeta: { lineCount: 3, charCount: 214 }, }, - { - index: 1, - id: "RateLimiter", - agent: "task", - agentSource: "bundled", - description: "Audit rate limiter", - task: "Inspect packages/server/src/auth/rate-limit.ts", - assignment: - "Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.", - exitCode: 0, - output: [ - "rate-limit.ts uses a fixed-window counter keyed by client IP.", - "429 responses set Retry-After (rate-limit.ts:57).", - "Gap: no per-account limit, so a botnet across IPs bypasses the cap.", - ].join("\n"), - stderr: "", - truncated: false, - durationMs: 38_500, - tokens: 54_800, - contextTokens: 19_700, - contextWindow: 200_000, - resolvedModel: "anthropic/claude-sonnet", - usage: { cost: { total: 0.1 } }, - outputMeta: { lineCount: 3, charCount: 198 }, - }, - { - index: 2, - id: "TokenRotation", - agent: "task", - agentSource: "bundled", - description: "Check token rotation", - task: "Trace refresh-token rotation in packages/server/src/auth/tokens.ts", - assignment: - "Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.", - exitCode: 0, - output: [ - "Refresh tokens rotate on every use (tokens.ts:120) and the old jti is revoked.", - "Reuse of a rotated token triggers full-family revocation — no reuse window found.", - ].join("\n"), - stderr: "", - truncated: false, - durationMs: 48_200, - tokens: 49_200, - contextTokens: 17_500, - contextWindow: 200_000, - resolvedModel: "anthropic/claude-sonnet", - usage: { cost: { total: 0.12 } }, - outputMeta: { lineCount: 2, charCount: 160 }, - }, ], - }, + } satisfies TaskToolDetails, }, errorResult: { isError: true, content: [ { type: "text", - text: "1 of 3 agents failed: RateLimiter.", + text: "Agent RateLimiter failed.", }, ], details: { projectAgentsDir: null, - totalDurationMs: 39_400, - usage: { cost: { total: 0.21 } }, + totalDurationMs: 9_800, + usage: fixtureUsage({ input: 10_900, output: 1_400 }, 0.1), results: [ { index: 0, - id: "AuthLoader", - agent: "task", - agentSource: "bundled", - description: "Load auth middleware", - task: "Read packages/server/src/auth/session.ts and middleware.ts", - assignment: - "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", - exitCode: 0, - output: "Session validation runs in middleware.ts:42 via verifySessionCookie().", - stderr: "", - truncated: false, - durationMs: 31_200, - tokens: 58_100, - contextTokens: 21_900, - contextWindow: 200_000, - resolvedModel: "anthropic/claude-sonnet", - usage: { cost: { total: 0.11 } }, - outputMeta: { lineCount: 1, charCount: 70 }, - }, - { - index: 1, id: "RateLimiter", agent: "task", agentSource: "bundled", @@ -184,15 +129,243 @@ export const agenticFixtures: Record = { truncated: false, durationMs: 9_800, tokens: 12_300, + requests: 3, contextTokens: 6_400, contextWindow: 200_000, resolvedModel: "anthropic/claude-sonnet", - usage: { cost: { total: 0.1 } }, + usage: fixtureUsage({ input: 10_900, output: 1_400 }, 0.1), error: "Subagent exited 1: target file packages/server/src/auth/rate-limit.ts does not exist.", outputMeta: { lineCount: 0, charCount: 0 }, }, ], - }, + } satisfies TaskToolDetails, + }, + }, + + // Resume: follow-up assignment into an existing (idle or parked) agent. + task_resume: { + label: "Task (resume)", + customRendered: true, + renderer: "task", + // Streaming: resume target known; the follow-up assignment still landing. + streamingArgs: { + resume: "AuthLoader", + assignment: "Follow up: does the sliding-expiration TODO affect", + }, + args: { + resume: "AuthLoader", + assignment: + "Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path? Document the answer.", + }, + result: { + content: [{ type: "text", text: "Agent AuthLoader completed." }], + details: { + projectAgentsDir: null, + totalDurationMs: 22_400, + usage: fixtureUsage({ input: 30_200, output: 4_100 }, 0.07), + results: [ + { + index: 0, + id: "AuthLoader", + agent: "task", + agentSource: "bundled", + task: "Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path?", + assignment: + "Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path? Document the answer.", + exitCode: 0, + output: + "No — refresh tokens bypass the sliding window: refreshSession() re-issues the cookie unconditionally (session.ts:131).", + stderr: "", + truncated: false, + durationMs: 19_700, + tokens: 34_300, + requests: 4, + contextTokens: 31_800, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: fixtureUsage({ input: 30_200, output: 4_100 }, 0.07), + outputMeta: { lineCount: 1, charCount: 118 }, + }, + ], + } satisfies TaskToolDetails, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: 'No agent "AuthLoader" to resume — it ran isolated and is not revivable. See history:// for the agent index.', + }, + ], + }, + }, + + irc: { + label: "IRC", + // Streaming: recipient known; the message body still arriving. + streamingArgs: { op: "send", to: "AuthLoader", message: "Are you still touching" }, + args: { + op: "send", + to: "AuthLoader", + message: "Are you still touching src/server/auth.ts? I need to add a 401 path.", + await: true, + }, + result: { + content: [ + { + type: "text", + text: [ + "Delivered to 1 peer(s):", + "- AuthLoader: revived", + "", + "Reply from AuthLoader:", + "Done with auth.ts — go ahead, just rebase past my session-store rename.", + ].join("\n"), + }, + ], + details: { + op: "send", + from: "Main", + to: "AuthLoader", + receipts: [{ to: "AuthLoader", outcome: "revived" }], + waited: { + id: "7181122334455667789", + from: "AuthLoader", + to: "Main", + body: "Done with auth.ts — go ahead, just rebase past my session-store rename.", + ts: FIXTURE_NOW - 5_000, + replyTo: "7181122334455667788", + }, + } satisfies IrcDetails, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: 'No recipients received the message.\n- RateLimiter: failed — unknown agent "RateLimiter"', + }, + ], + details: { + op: "send", + from: "Main", + to: "RateLimiter", + receipts: [{ to: "RateLimiter", outcome: "failed", error: 'unknown agent "RateLimiter"' }], + } satisfies IrcDetails, + }, + }, + + irc_wait: { + label: "IRC (wait)", + customRendered: true, + renderer: "irc", + streamingArgs: { op: "wait", from: "AuthLoader" }, + args: { op: "wait", from: "AuthLoader", timeoutMs: 60_000 }, + result: { + content: [ + { + type: "text", + text: "[7181122334455667790] AuthLoader: session-store rename is merged; auth.ts is yours.", + }, + ], + details: { + op: "wait", + from: "Main", + waited: { + id: "7181122334455667790", + from: "AuthLoader", + to: "Main", + body: "session-store rename is merged; auth.ts is yours.", + ts: FIXTURE_NOW - 30_000, + }, + } satisfies IrcDetails, + }, + }, + + irc_inbox: { + label: "IRC (inbox)", + customRendered: true, + renderer: "irc", + streamingArgs: { op: "inbox" }, + args: { op: "inbox", peek: true }, + result: { + content: [ + { + type: "text", + text: [ + "2 unread message(s):", + "- [7181122334455667791] AuthLoader: hub table reads unreadCount — ping me when the bus lands.", + "- [7181122334455667792] RateLimiter (reply to 7181122334455667791): bus is in; receipts carry outcome.", + ].join("\n"), + }, + ], + details: { + op: "inbox", + from: "Main", + inbox: [ + { + id: "7181122334455667791", + from: "AuthLoader", + to: "Main", + body: "hub table reads unreadCount — ping me when the bus lands.", + ts: FIXTURE_NOW - 4 * 60_000, + }, + { + id: "7181122334455667792", + from: "RateLimiter", + to: "Main", + body: "bus is in; receipts carry outcome.", + ts: FIXTURE_NOW - 60_000, + replyTo: "7181122334455667791", + }, + ], + } satisfies IrcDetails, + }, + }, + + irc_list: { + label: "IRC (list)", + customRendered: true, + renderer: "irc", + streamingArgs: { op: "list" }, + args: { op: "list" }, + result: { + content: [ + { + type: "text", + text: [ + "2 peer(s):", + "- AuthLoader [task · sub · idle] — parent Main, active 2m ago", + "- RateLimiter [task · sub · parked] — unread 2, parent Main, active 12m ago", + "", + "Parked agents are revived automatically when you message them.", + ].join("\n"), + }, + ], + details: { + op: "list", + from: "Main", + peers: [ + { + id: "AuthLoader", + displayName: "task", + kind: "sub", + status: "idle", + parentId: "Main", + unread: 0, + lastActivity: FIXTURE_NOW - 2 * 60_000, + }, + { + id: "RateLimiter", + displayName: "task", + kind: "sub", + status: "parked", + parentId: "Main", + unread: 2, + lastActivity: FIXTURE_NOW - 12 * 60_000, + }, + ], + } satisfies IrcDetails, }, }, diff --git a/packages/coding-agent/src/cli/gallery-fixtures/types.ts b/packages/coding-agent/src/cli/gallery-fixtures/types.ts index da4b9b2e4..cdf935e16 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/types.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/types.ts @@ -36,6 +36,11 @@ export interface GalleryFixture { * real one keeps the gallery honest for these tools. */ customRendered?: boolean; + /** + * Renderer-registry key to use when the fixture key is a variant of a tool + * (e.g. `task_resume` → `task`). Defaults to the fixture key. + */ + renderer?: string; /** * Arguments shown during the streaming state — a partial view of {@link args} * as if the tool-call JSON were still arriving. May include `__partialJson` diff --git a/packages/coding-agent/src/commit/agentic/tools/analyze-file.ts b/packages/coding-agent/src/commit/agentic/tools/analyze-file.ts index 78f0e7b2b..7d09113e7 100644 --- a/packages/coding-agent/src/commit/agentic/tools/analyze-file.ts +++ b/packages/coding-agent/src/commit/agentic/tools/analyze-file.ts @@ -59,29 +59,46 @@ export function createAnalyzeFileTool(options: { label: "Analyze Files", description: "Spawn quick_task agents to analyze files.", parameters: analyzeFileSchema, - async execute(toolCallId, params, onUpdate, ctx, signal) { + async execute(toolCallId, params, _onUpdate, ctx, signal) { const toolSession = buildToolSession(ctx, options); + // The hand-built ToolSession carries no asyncJobManager, so every + // execute() below takes the task tool's sync fallback and resolves + // with the subagent's result inline — exactly what this flow needs. + // The tool's session semaphore bounds the parallel fan-out. const taskTool = await TaskTool.create(toolSession); const numstat = options.state.overview?.numstat ?? []; - const tasks = params.files.map((file, index) => { - const relatedFiles = formatRelatedFiles(params.files, file, numstat); - const assignment = prompt.render(analyzeFilePrompt, { - file, - goal: params.goal, - related_files: relatedFiles, - }); - return { - id: `AnalyzeFile${index + 1}`, - description: `Analyze ${file}`, - assignment, - }; - }); - const taskParams: TaskParams = { - agent: "quick_task", - schema: JSON.stringify(analyzeFileOutputSchema), - tasks, + const schema = JSON.stringify(analyzeFileOutputSchema); + const analyses = await Promise.all( + params.files.map((file, index) => { + const relatedFiles = formatRelatedFiles(params.files, file, numstat); + const assignment = prompt.render(analyzeFilePrompt, { + file, + goal: params.goal, + related_files: relatedFiles, + }); + const taskParams: TaskParams = { + agent: "quick_task", + id: `AnalyzeFile${index + 1}`, + description: `Analyze ${file}`, + assignment, + schema, + }; + return taskTool.execute(`${toolCallId}-${index + 1}`, taskParams, signal); + }), + ); + const results = analyses.flatMap(analysis => analysis.details?.results ?? []); + const text = analyses + .map(analysis => analysis.content.find(part => part.type === "text")?.text ?? "") + .filter(Boolean) + .join("\n\n"); + return { + content: [{ type: "text", text: text || "(no output)" }], + details: { + projectAgentsDir: null, + results, + totalDurationMs: analyses.reduce((sum, analysis) => sum + (analysis.details?.totalDurationMs ?? 0), 0), + }, }; - return taskTool.execute(toolCallId, taskParams, signal, onUpdate); }, }; } diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 3b2edc579..3b20c0387 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -2856,6 +2856,34 @@ export const SETTINGS_SCHEMA = { }, }, + "task.agentIdleTtlMs": { + type: "number", + default: 420_000, + ui: { + tab: "tasks", + label: "Agent Idle TTL", + description: + "How long an idle subagent stays live in memory before being parked to disk (ms). Parked agents are revived automatically when messaged or resumed. 0 keeps idle agents live until exit.", + }, + }, + + "task.softRequestBudget": { + type: "number", + default: 90, + ui: { + tab: "tasks", + label: "Soft Subagent Request Budget", + description: + "Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/quick_task agents use a lower built-in budget.", + options: [ + { value: "0", label: "Disabled" }, + { value: "40", label: "40 requests" }, + { value: "90", label: "90 requests", description: "Default" }, + { value: "150", label: "150 requests" }, + ], + }, + }, + "task.disabledAgents": { type: "array", default: [] as string[], diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 588fd1b5a..1c63b3431 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -99,6 +99,7 @@ function singleResult(options: ExecutorOptions, overrides: Partial truncated: false, durationMs: 1, tokens: 0, + requests: 0, ...overrides, }; } @@ -541,6 +542,7 @@ describe("agent() through eval runtimes", () => { recentOutput: [], toolCount: 0, tokens: 0, + requests: 0, cost: 0, durationMs: 0, ...overrides, @@ -673,6 +675,7 @@ describe("agent() through eval runtimes", () => { recentOutput: [], toolCount: i, tokens: 0, + requests: 0, cost: 0, durationMs: i * 10, }); diff --git a/packages/coding-agent/src/prompts/system/orchestrate-notice.md b/packages/coding-agent/src/prompts/system/orchestrate-notice.md index c8086fbb4..cd2754909 100644 --- a/packages/coding-agent/src/prompts/system/orchestrate-notice.md +++ b/packages/coding-agent/src/prompts/system/orchestrate-notice.md @@ -8,7 +8,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor 1. **NEVER yield until everything is closed.** A phase finishing is *not* a yield point — launch the next phase in the same turn. Stop only when every requested item is verifiably done, or you hit a concrete [blocked] state that genuinely requires the user. 2. **Enumerate the full surface before dispatching.** If the request references audits, plans, checklists, phase lists, or file lists, expand them into a flat set of items in `todo`. "Most of them" or "the important ones" is failure. Re-read the source documents — NEVER work from memory. -3. **Parallelize maximally; NEVER launch a one-off task.** Every set of edits with disjoint file scope MUST ship as one `task` batch — fan the work as wide as it decomposes. A single-task batch for divisible work is a failure: split it. If you are about to dispatch exactly one subagent, stop — either there is more to run alongside it (find it and batch them) or the change is small enough to make inline yourself (do it). Serialize only when one subagent produces a contract (types, schema, shared module) the next consumes — and state the dependency when you do. +3. **Parallelize maximally; NEVER launch a one-off task.** Every set of edits with disjoint file scope MUST ship as parallel `task` calls in one message — fan the work as wide as it decomposes. Dispatching divisible work one call at a time, serially, is a failure: split it and dispatch together. If you are about to dispatch exactly one subagent, stop — either there is more to run alongside it (find it and dispatch them together) or the change is small enough to make inline yourself (do it). Serialize only when one subagent produces a contract (types, schema, shared module) the next consumes — and state the dependency when you do. 4. **Each `task` assignment is self-contained.** Subagents have no shared context. Spell out: target files (≤3–5 explicit paths, no globs), the change with APIs and patterns, edge cases, and observable acceptance criteria. NEVER assume they read the same plan you did. 5. **Verify after every phase before launching the next.** Run the appropriate gate: `bun check` for types, package-scoped `bun test` for behavior, `lsp diagnostics` for changed files. If a phase introduced breakage, dispatch fix-up subagents *before* moving on. NEVER declare a phase done on a red tree. 6. **Commit policy.** If the request asks for commits or the repo workflow expects them, commit after each green phase with a focused message. NEVER commit a red tree. NEVER commit work the user did not ask to commit. @@ -21,7 +21,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor 1. **Ingest.** Read every referenced file (audits, plans, prior agent output, current branch state). Run `git status` to see uncommitted changes. 2. **Plan.** Materialize the full work surface in `todo` as ordered phases. Within each phase, list the parallelizable units. -3. **Dispatch phase.** Launch all parallel `task` subagents in one call. Wait for the batch. +3. **Dispatch phase.** Launch all parallel `task` subagents in one message, then collect every result (async results / `job poll`) before moving on. 4. **Verify phase.** Run the gates. On failure, dispatch fix-up subagents and re-verify. Do not advance with a red gate. 5. **Commit phase** (if applicable). Focused message naming the phase. 6. **Advance.** Mark the phase done in `todo`, immediately start the next phase. No summary message between phases — keep going. diff --git a/packages/coding-agent/src/prompts/tools/task-summary.md b/packages/coding-agent/src/prompts/tools/task-summary.md index b6a945351..21f21d9b6 100644 --- a/packages/coding-agent/src/prompts/tools/task-summary.md +++ b/packages/coding-agent/src/prompts/tools/task-summary.md @@ -1,28 +1,17 @@ - -
{{successCount}}/{{totalCount}} succeeded{{#if hasCancelledNote}} ({{cancelledCount}} cancelled){{/if}} [{{duration}}]
- -{{#each summaries}} - -{{status}} + {{#if meta}}{{/if}} {{#if truncated}} - + {{preview}} {{else}} - + {{preview}} - + {{/if}} - -{{#unless @last}} ---- -{{/unless}} -{{/each}} - {{#if mergeSummary}} {{mergeSummary}} {{/if}} -
+ diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index eb2e8cd83..d567568c2 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -1,43 +1,37 @@ -Launches subagents to parallelize workflows. +Spawns ONE subagent per call to work in the background, or resumes an existing one. -{{#if asyncEnabled}} -- Results are delivered automatically when complete. -- The tool result lists the assigned task ids (e.g. `AuthLoader`) — those are the live agent ids. +- Spawning is non-blocking: the call returns immediately with the agent id and a job id; the result is delivered automatically when the agent yields. +- Parallelism = multiple `task` calls in one assistant message. Concurrency is bounded at {{MAX_CONCURRENCY}} running subagents per session. +- If genuinely blocked on a result, wait with `job poll`; otherwise keep working. `job cancel` terminates a task and **cannot carry a message** — only for stalled/abandoned work. {{#if ircEnabled}} -- Coordinate with running tasks via `irc` using those ids. `job cancel` terminates a task and **cannot carry a message** — only use it for stalled/abandoned work. -- If genuinely blocked on completion, wait with `job poll`; otherwise keep working. -{{else}} -- If genuinely blocked on completion, wait with `job poll`; otherwise keep working. -- Use `job list` to snapshot manager state; `cancel: [id]` only to actually stop a stuck task. -{{/if}} +- Coordinate with running agents via `irc` using their ids. Agents reach you and their siblings live the same way. {{/if}} -{{#if ircEnabled}} -Subagents have no conversation history, but they can reach you and their siblings live via the `irc` tool. Front-load every fact, file path, and direction they need in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}. -{{else}} -Subagents have no conversation history. Every fact, file path, and direction they need MUST be explicit in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}. -{{/if}} + +- Finished agents stay alive: `idle` first, then `parked` after a TTL — both remain addressable and revivable. +- `resume: ""` revives an idle/parked agent and runs a follow-up assignment in its existing session. **Prefer resuming an agent that already holds the relevant context over spawning fresh**{{#if ircEnabled}} — check `irc` op:"list" for candidates{{/if}}. +- `history://` is the agent's transcript; `agent://` its latest output artifact. + -- `agent`: agent type for all tasks -- `tasks`: tasks to execute in parallel - - `.id`: CamelCase, ≤32 chars - - `.description`: UI label only — subagent never sees it - - `.assignment`: complete self-contained instructions; one-liners and missing acceptance criteria are PROHIBITED -{{#if contextEnabled}}- `context`: shared background prepended to every assignment; session-specific only{{/if}} +- `agent`: agent type to spawn; omit when `resume` is set +- `resume`: existing agent id — continue that agent instead of spawning (cannot combine with `agent` or `isolated`) +- `id`: stable agent id, CamelCase, ≤32 chars; generated when omitted +- `description`: UI label only — subagent never sees it +- `assignment`: complete self-contained instructions; one-liners and missing acceptance criteria are PROHIBITED {{#if customSchemaEnabled}}- `schema`: JTD schema for expected structured output (do not put format rules in assignments){{/if}} -{{#if isolationEnabled}}- `isolated`: run in isolated env; use when tasks edit overlapping files{{/if}} +{{#if isolationEnabled}}- `isolated`: run in isolated env; returns patches. Isolated agents are NOT resumable{{/if}} -- **Maximize batch width.** Spawn the widest parallel set the work decomposes into. NEVER spawn a single-task batch for divisible work, or defer work that could have been concurrent. -- **Subagents do not verify, lint, or format.** Every assignment MUST instruct the subagent to skip all gates, formatters, and project-wide build/test/lint. You run them once at the end across the union of changed files — avoids redundant runs and racing formatter passes. +- **Maximize fan-out.** Issue the widest set of parallel `task` calls the work decomposes into. NEVER serialize work that could run concurrently. +- **Subagents do not verify, lint, or format.** Every assignment MUST instruct the subagent to skip all gates, formatters, and project-wide build/test/lint. You run them once at the end across the union of changed files. - No globs, no "update all", no package-wide scope. Fan out. - NEVER slow down or serialize because tasks might overlap on some files. Agents resolve collisions among themselves in real time. -- Pass large payloads via `local://` URIs, not inline. {{#if contextEnabled}} (other than the context){{/if}} -{{#if contextEnabled}}- Put shared constraints in `context` once; do not duplicate across assignments.{{/if}} +- Subagents have no conversation history. Every fact, file path, and direction they need MUST be explicit in the `assignment`. +- **Shared background**: write it ONCE to a `local://` file (e.g. `local://ctx.md`) and reference that path in each assignment. Pass large payloads via `local://` URIs, not inline. - Prefer agents that investigate **and** edit in one pass; only spin a read-only discovery step when affected files are genuinely unknown. -- **Read-only agents**: Agents tagged READ-ONLY (e.g. `explore`) have no edit/write/command tools. NEVER hand them an assignment that requires changing files or running commands — they cannot do it and the turn is wasted. Use them to investigate and report back; do the edits yourself or delegate to a writing agent (`task`, `oracle`, `designer`). +- **Read-only agents**: Agents tagged READ-ONLY (e.g. `explore`) have no edit/write/command tools. NEVER hand them an assignment that requires changing files or running commands. Use them to investigate and report back; do the edits yourself or delegate to a writing agent (`task`, `oracle`, `designer`). - **No reasoning offload**: NEVER offload reasoning, analysis, design, or decision-making to `quick_task` or `explore` — they run minimal-effort / small models for mechanical lookups and data collection only. Keep judgment and synthesis in your own context; delegate hard thinking to `task`, `plan`, or `oracle`. @@ -51,16 +45,9 @@ Test: can task B run correctly without seeing A's output? If no, sequence A → Sequential when one task produces a contract (types, API, schema, core module) the other consumes. Parallel when tasks touch disjoint files or are independent refactors/tests. {{/if}} +Sequenced follow-ups SHOULD `resume` the agent that produced the prerequisite — it already holds the context. -{{#if contextEnabled}} - -# Goal ← one sentence: what the batch accomplishes -# Constraints ← MUST/NEVER rules and session decisions -# Contract ← exact types/signatures if tasks share an interface - -{{/if}} - # Target ← exact files and symbols; explicit non-goals # Change ← step-by-step add/remove/rename; APIs and patterns diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index a0ebd42c1..dda69eeef 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -7,6 +7,7 @@ import path from "node:path"; import type { AgentEvent, AgentIdentity, AgentTelemetryConfig, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { recordHandoff, resolveTelemetry } from "@oh-my-pi/pi-agent-core"; +import type { Usage } from "@oh-my-pi/pi-ai"; import { logger, prompt, untilAborted } from "@oh-my-pi/pi-utils"; import type { Rule } from "../capability/rule"; import { ModelRegistry } from "../config/model-registry"; @@ -26,8 +27,9 @@ import type { MCPManager } from "../mcp/manager"; import type { MnemopiSessionState } from "../mnemopi/state"; import subagentSystemPromptTemplate from "../prompts/system/subagent-system-prompt.md" with { type: "text" }; import submitReminderTemplate from "../prompts/system/subagent-yield-reminder.md" with { type: "text" }; +import { AgentLifecycleManager } from "../registry/agent-lifecycle"; import { AgentRegistry } from "../registry/agent-registry"; -import { createAgentSession, discoverAuthStorage } from "../sdk"; +import { type CreateAgentSessionOptions, createAgentSession, discoverAuthStorage } from "../sdk"; import type { AgentSession, AgentSessionEvent } from "../session/agent-session"; import type { ArtifactManager } from "../session/artifacts"; import type { AuthStorage } from "../session/auth-storage"; @@ -63,6 +65,30 @@ import { const MCP_CALL_TIMEOUT_MS = 60_000; +/** + * Soft per-agent request budgets (assistant requests per run). When a subagent + * crosses its budget it receives ONE steering notice asking it to wrap up; at + * 1.5x the budget the run is aborted gracefully so partial output is salvaged. + * The `default` key applies to agents without an explicit entry and can be + * overridden via the `task.softRequestBudget` setting (0 disables the guard). + */ +export const SOFT_REQUEST_BUDGET: Record = { + explore: 40, + quick_task: 40, + default: 90, +}; + +/** Steering notice injected once when a subagent crosses its soft request budget. */ +export function buildBudgetNotice(requests: number): string { + return `[budget notice] You have used ${requests} requests in this run. Wrap up now: finish the current step and yield your final report.`; +} + +/** Flatten whitespace and clip salvage text for the cancelled-child summary line. */ +function formatSalvageSnippet(text: string, maxLength = 500): string { + const flattened = text.replace(/\s+/g, " ").trim(); + return flattened.length > maxLength ? `${flattened.slice(0, maxLength - 1)}…` : flattened; +} + /** Agent event types to forward for progress tracking. */ const agentEventTypes = new Set([ "agent_start", @@ -94,9 +120,13 @@ function normalizeModelPatterns(value: string | string[] | undefined): string[] function renderIrcPeerRoster(selfId: string): string { const peers = AgentRegistry.global() .list() - .filter(ref => ref.id !== selfId && (ref.status === "running" || ref.status === "idle")); - if (peers.length === 0) return "- (no other live agents)"; - return peers.map(peer => `- \`${peer.id}\` — ${peer.displayName} (${peer.kind}, ${peer.status})`).join("\n"); + .filter(ref => ref.id !== selfId && ref.status !== "aborted"); + if (peers.length === 0) return "- (no other agents)"; + const lines = peers.map(peer => `- \`${peer.id}\` — ${peer.displayName} (${peer.kind}, ${peer.status})`); + if (peers.some(peer => peer.status === "idle" || peer.status === "parked")) { + lines.push("Idle/parked peers are not gone: messaging them wakes (or revives) them."); + } + return lines.join("\n"); } function withAbortTimeout(promise: Promise, timeoutMs: number, signal?: AbortSignal): Promise { @@ -152,7 +182,6 @@ export interface ExecutorOptions { agent: AgentDefinition; task: string; assignment?: string; - context?: string; /** * The session's active overall plan, handed off so subagents spawned during * plan execution share the same plan context as the main agent. Omitted when @@ -186,8 +215,6 @@ export interface ExecutorOptions { sessionFile?: string | null; persistArtifacts?: boolean; artifactsDir?: string; - /** Path to parent conversation context file */ - contextFile?: string; eventBus?: EventBus; contextFiles?: ContextFileEntry[]; skills?: Skill[]; @@ -611,28 +638,67 @@ export function createSubagentSettings( }); } +type AbortReason = "signal" | "terminate" | "timeout" | "budget"; + +/** Inputs for the shared run monitor used by both fresh spawns and resumes. */ +interface RunMonitorArgs { + index: number; + id: string; + agent: AgentDefinition; + task: string; + assignment?: string; + description?: string; + modelOverride?: string | string[]; + signal?: AbortSignal; + onProgress?: (progress: AgentProgress) => void; + eventBus?: EventBus; + parentToolCallId?: string; + sessionFile?: string; + /** Soft assistant-request budget; 0 disables the guard. */ + softRequestBudget: number; + /** Wall-clock cap in ms; 0 disables the timer. */ + maxRuntimeMs: number; +} + /** - * Run a single agent in-process. + * The run-monitoring core shared by {@link runSubprocess} and + * {@link resumeSubprocess}: progress tracking, event processing, abort/budget + * machinery, usage accumulation, and output capture for one assignment run. */ -export async function runSubprocess(options: ExecutorOptions): Promise { - const { - cwd, - agent, - task, - assignment, - index, - id, - worktree, - modelOverride, - thinkingLevel, - outputSchema, - enableLsp, - signal, - onProgress, - } = options; +interface SubagentRunMonitor { + readonly progress: AgentProgress; + /** Fires when the run was asked to stop (caller signal, timeout, budget, terminate). */ + readonly abortSignal: AbortSignal; + readonly accumulatedUsage: Usage; + hasUsage(): boolean; + yieldCalled(): boolean; + runtimeLimitExceeded(): boolean; + /** True when the abort carries a precise external reason (signal / wall-clock / budget). */ + hasExplicitAbortReason(): boolean; + /** Whether the (attempted) abort counts as a cancelled run rather than an internal failure. */ + isAbortedRun(): boolean; + requestAbort(reason: AbortReason): void; + resolveSignalAbortReason(): string; + resolveAbortReasonText(): string; + setActiveSession(session: AgentSession | null): void; + /** Return and clear the active session reference. */ + takeActiveSession(): AgentSession | null; + /** Subscribe the monitor to a session's events. Returns the unsubscribe function. */ + attach(session: AgentSession): () => void; + /** Best-effort capture of the last assistant text for cancelled-run salvage. */ + captureSalvage(session: AgentSession): void; + lastAssistantSalvageText(): string | undefined; + /** Final raw output: end-of-run assistant text when available, else accumulated chunks. */ + rawOutput(): string; + scheduleProgress(flush?: boolean): void; + /** Stop processing events and clear listeners/timers. Call once the run settled. */ + finish(): void; +} + +function createSubagentRunMonitor(args: RunMonitorArgs): SubagentRunMonitor { + const { index, id, agent, task, assignment, signal, onProgress, softRequestBudget, maxRuntimeMs } = args; const startTime = Date.now(); - // Initialize progress const progress: AgentProgress = { index, id, @@ -641,109 +707,23 @@ export async function runSubprocess(options: ExecutorOptions): Promise= 0 && childDepth >= maxRecursionDepth; - - // Add tools if specified - let toolNames: string[] | undefined; - if (agent.tools && agent.tools.length > 0) { - toolNames = agent.tools; - // Auto-include task tool if spawns defined but task not in tools - if (agent.spawns !== undefined && !toolNames.includes("task") && !atMaxDepth) { - toolNames = [...toolNames, "task"]; - } - } - - if (atMaxDepth && toolNames?.includes("task")) { - toolNames = toolNames.filter(name => name !== "task"); - } - // IRC is always available; the COOP prompt section advertises it, so a restricted - // whitelist must still carry `irc` for the subagent to actually use it. - if (toolNames && !toolNames.includes("irc")) { - toolNames = [...toolNames, "irc"]; - } - if (toolNames?.includes("exec")) { - const allowEvalPy = settings.get("eval.py") ?? true; - const allowEvalJs = settings.get("eval.js") ?? true; - const expanded = toolNames.filter(name => name !== "exec"); - if (allowEvalPy || allowEvalJs) expanded.push("eval"); - expanded.push("bash"); - toolNames = Array.from(new Set(expanded)); - } - - const modelPatterns = normalizeModelPatterns(modelOverride ?? agent.model); - const sessionFile = subtaskSessionFile ?? null; - const spawnsEnv = atMaxDepth - ? "" - : agent.spawns === undefined - ? "" - : agent.spawns === "*" - ? "*" - : agent.spawns.join(","); - - const lspEnabled = enableLsp ?? true; - const ircEnabled = subagentSettings.get("irc.enabled") === true; - const contextFileForPrompt = ircEnabled ? undefined : options.contextFile; - const skipPythonPreflight = Array.isArray(toolNames) && !toolNames.includes("eval"); - const outputChunks: string[] = []; const finalOutputChunks: string[] = []; const RECENT_OUTPUT_TAIL_BYTES = 8 * 1024; let recentOutputTail = ""; - let stderr = ""; let resolved = false; - type AbortReason = "signal" | "terminate" | "timeout"; let abortSent = false; let abortReason: AbortReason | undefined; let runtimeLimitExceeded = false; @@ -752,11 +732,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise void) | null = null; let yieldCalled = false; // Accumulate usage incrementally from message_end events (no memory for streaming events) - const accumulatedUsage = { + const accumulatedUsage: Usage = { input: 0, output: 0, cacheRead: 0, @@ -765,11 +744,17 @@ export async function runSubprocess(options: ExecutorOptions): Promise { if (reason === "timeout") { runtimeLimitExceeded = true; } + if (reason === "budget") { + budgetLimitExceeded = true; + } if (abortSent) { if (reason === "signal" && abortReason !== "signal" && abortReason !== "timeout") { abortReason = "signal"; @@ -786,11 +771,14 @@ export async function runSubprocess(options: ExecutorOptions): Promise { - if (!resolved) requestAbort("signal"); - }; if (signal) { - signal.addEventListener("abort", onAbort, { once: true, signal: listenerSignal }); + signal.addEventListener( + "abort", + () => { + if (!resolved) requestAbort("signal"); + }, + { once: true, signal: listenerSignal }, + ); } // Wall-clock hard limit. Defense-in-depth for the case where a provider stream @@ -826,6 +814,9 @@ export async function runSubprocess(options: ExecutorOptions): Promise { progress.durationMs = Date.now() - startTime; onProgress?.({ ...progress }); - if (options.eventBus) { - options.eventBus.emit(TASK_SUBAGENT_PROGRESS_CHANNEL, { + if (args.eventBus) { + args.eventBus.emit(TASK_SUBAGENT_PROGRESS_CHANNEL, { index, agent: agent.name, agentSource: agent.source, task, - parentToolCallId: options.parentToolCallId, + parentToolCallId: args.parentToolCallId, assignment, progress: { ...progress }, - sessionFile: subtaskSessionFile, + sessionFile: args.sessionFile, }); } lastProgressEmitMs = Date.now(); @@ -925,8 +916,8 @@ export async function runSubprocess(options: ExecutorOptions): Promise { - if (!options.eventBus) return; - options.eventBus.emit(TASK_SUBAGENT_EVENT_CHANNEL, { + if (!args.eventBus) return; + args.eventBus.emit(TASK_SUBAGENT_EVENT_CHANNEL, { id, event, }); @@ -1078,6 +1069,26 @@ export async function runSubprocess(options: ExecutorOptions): Promise 0 && !abortSent) { + if (progress.requests >= softRequestBudget * 1.5) { + requestAbort("budget"); + } else if (!budgetSteerSent && progress.requests >= softRequestBudget) { + budgetSteerSent = true; + const steerSession = activeSession; + if (steerSession) { + void steerSession + .sendUserMessage(buildBudgetNotice(progress.requests), { deliverAs: "steer" }) + .catch(err => { + logger.warn("Subagent budget steer failed", { + error: err instanceof Error ? err.message : String(err), + }); + }); + } + } + } + } if (role === "assistant") { const messageContent = getMessageContent(event.message) || (event as AgentEvent & { content?: unknown }).content; @@ -1147,6 +1158,543 @@ export async function runSubprocess(options: ExecutorOptions): Promise void) => + session.subscribe(event => { + emitSubagentEvent(event); + if (event.type === "auto_retry_start") { + progress.retryState = { + attempt: event.attempt, + maxAttempts: event.maxAttempts, + delayMs: event.delayMs, + errorMessage: event.errorMessage, + startedAtMs: Date.now(), + }; + progress.retryFailure = undefined; + scheduleProgress(true); + return; + } + if (event.type === "auto_retry_end") { + const attempt = progress.retryState?.attempt ?? event.attempt; + progress.retryState = undefined; + if (!event.success) { + progress.retryFailure = { + attempt, + errorMessage: event.finalError ?? "Auto-retry failed", + }; + } + scheduleProgress(true); + return; + } + if (isAgentEvent(event)) { + try { + processEvent(event); + } catch (err) { + logger.error("Subagent event processing failed", { + error: err instanceof Error ? err.message : String(err), + }); + requestAbort("terminate"); + } + } + }); + + const captureSalvage = (session: AgentSession): void => { + // Best-effort salvage: capture the last assistant text so + // cancelled/aborted children can surface "last activity" instead of + // "(no output)". + try { + const lastContent = session.getLastAssistantMessage()?.content; + if (Array.isArray(lastContent)) { + const text = lastContent + .map(block => (block.type === "text" && typeof block.text === "string" ? block.text : "")) + .filter(Boolean) + .join("\n"); + if (text.trim()) { + lastAssistantSalvageText = text; + } + } + } catch { + // Salvage is best-effort; partial sessions may not implement it + } + }; + + return { + progress, + abortSignal, + accumulatedUsage, + hasUsage: () => hasUsage, + yieldCalled: () => yieldCalled, + runtimeLimitExceeded: () => runtimeLimitExceeded, + hasExplicitAbortReason: () => abortReason === "signal" || runtimeLimitExceeded || budgetLimitExceeded, + isAbortedRun: () => + abortReason === "signal" || runtimeLimitExceeded || budgetLimitExceeded || abortReason === undefined, + requestAbort, + resolveSignalAbortReason, + resolveAbortReasonText, + setActiveSession: session => { + activeSession = session; + }, + takeActiveSession: () => { + const session = activeSession; + activeSession = null; + return session; + }, + attach, + captureSalvage, + lastAssistantSalvageText: () => lastAssistantSalvageText, + rawOutput: () => (finalOutputChunks.length > 0 ? finalOutputChunks.join("") : outputChunks.join("")), + scheduleProgress, + finish: () => { + resolved = true; + listenerController.abort(); + if (runtimeTimeoutId !== undefined) { + clearTimeout(runtimeTimeoutId); + runtimeTimeoutId = undefined; + } + if (progressTimeoutId) { + clearTimeout(progressTimeoutId); + progressTimeoutId = null; + } + }, + }; +} + +interface DriveOutcome { + exitCode: number; + error?: string; + aborted: boolean; + abortReasonText?: string; +} + +const MAX_YIELD_RETRIES = 3; + +/** + * Drive one assignment through a live session: send the prompt, wait for idle, + * remind the agent to `yield` (up to {@link MAX_YIELD_RETRIES} times), then + * classify the terminal assistant state. Shared by spawn and resume paths. + */ +async function driveSessionToYield( + session: AgentSession, + monitor: SubagentRunMonitor, + task: string, +): Promise { + const abortSignal = monitor.abortSignal; + let exitCode = 0; + let error: string | undefined; + let aborted = false; + let abortReasonText: string | undefined; + const checkAbort = () => { + if (abortSignal.aborted) { + aborted = monitor.isAbortedRun(); + if (aborted) { + abortReasonText ??= monitor.resolveAbortReasonText(); + } + exitCode = 1; + throw new ToolAbortError(); + } + }; + const awaitAbortable = async (promise: Promise): Promise => { + checkAbort(); + const { promise: abortPromise, reject } = Promise.withResolvers(); + const onAbort = () => { + try { + checkAbort(); + } catch (err) { + reject(err); + } + }; + abortSignal.addEventListener("abort", onAbort, { once: true }); + try { + return await Promise.race([promise, abortPromise]); + } finally { + abortSignal.removeEventListener("abort", onAbort); + } + }; + + try { + await awaitAbortable(session.prompt(task, { attribution: "agent" })); + await awaitAbortable(session.waitForIdle()); + + const reminderToolChoice = buildNamedToolChoice("yield", session.model); + + let retryCount = 0; + while (!monitor.yieldCalled() && retryCount < MAX_YIELD_RETRIES && !abortSignal.aborted) { + // Skip reminders when the model returned a terminal error (e.g. + // rate-limit cap hit, auth failure). Re-prompting would just + // hit the same wall, multiplying the failure noise without + // any chance of producing a yield. + const lastBeforeReminder = session.getLastAssistantMessage(); + if (lastBeforeReminder?.stopReason === "error") break; + try { + retryCount++; + const reminder = prompt.render(submitReminderTemplate, { + retryCount, + maxRetries: MAX_YIELD_RETRIES, + }); + + const isFinalRetry = retryCount >= MAX_YIELD_RETRIES; + await awaitAbortable( + session.prompt(reminder, { + attribution: "agent", + synthetic: true, + ...(isFinalRetry && reminderToolChoice ? { toolChoice: reminderToolChoice } : {}), + }), + ); + await awaitAbortable(session.waitForIdle()); + } catch (err) { + if (abortSignal.aborted || err instanceof ToolAbortError) { + // Benign control-flow exit — user cancel (^C) or compaction aborting + // pending operations both surface here as ToolAbortError. The outer + // catch and finally already mark the run aborted; logging at ERROR + // would spam operator dashboards with non-failures. + logger.debug("Subagent prompt aborted"); + } else { + logger.error("Subagent prompt failed", { + error: err instanceof Error ? err.message : String(err), + }); + } + } + } + + await awaitAbortable(session.waitForIdle()); + + const lastAssistant = session.getLastAssistantMessage(); + if (lastAssistant) { + if (lastAssistant.stopReason === "aborted") { + aborted = monitor.isAbortedRun(); + if (aborted) { + // A real caller signal or the wall-clock timer carries a precise + // reason (signal.reason / "runtime limit exceeded"). An internal + // turn abort does NOT — prefer the assistant message's own + // errorMessage ("Request was aborted" or a specific stream error) + // over the misleading "Cancelled by caller". + abortReasonText ??= monitor.hasExplicitAbortReason() + ? monitor.resolveAbortReasonText() + : lastAssistant.errorMessage?.trim() || monitor.resolveAbortReasonText(); + } + exitCode = 1; + } else if (lastAssistant.stopReason === "error") { + exitCode = 1; + error ??= lastAssistant.errorMessage || "Subagent failed"; + } + } + } catch (err) { + exitCode = 1; + if (!abortSignal.aborted) { + error = err instanceof Error ? err.stack || err.message : String(err); + } + } finally { + if (abortSignal.aborted) { + aborted = monitor.isAbortedRun(); + if (aborted) { + abortReasonText ??= monitor.resolveAbortReasonText(); + } + if (exitCode === 0) exitCode = 1; + } + } + + return { exitCode, error, aborted, abortReasonText }; +} + +interface FinalizeRunArgs { + monitor: SubagentRunMonitor; + done: { exitCode: number; error?: string; aborted?: boolean; abortReason?: string; durationMs: number }; + index: number; + id: string; + agent: AgentDefinition; + task: string; + assignment?: string; + description?: string; + modelOverride?: string | string[]; + outputSchema?: unknown; + signal?: AbortSignal; + artifactsDir?: string; + eventBus?: EventBus; + parentToolCallId?: string; + sessionFile?: string; + startTime: number; +} + +/** + * Turn a settled run into a {@link SingleResult}: resolve the yield payload via + * {@link finalizeSubprocessOutput}, salvage cancelled-run output, write the + * `.md` output artifact, flush final progress, and emit the lifecycle end + * event. Shared by spawn and resume paths. + */ +async function finalizeRunResult(args: FinalizeRunArgs): Promise { + const { monitor, done, index, id, agent, task, assignment, signal, modelOverride } = args; + const progress = monitor.progress; + let exitCode = done.exitCode; + let stderr = done.error ?? ""; + + // Use final output if available, otherwise accumulated output + let rawOutput = monitor.rawOutput(); + const yieldItems = progress.extractedToolData?.yield as YieldItem[] | undefined; + const reportFindingDetails = progress.extractedToolData?.report_finding as ReportFindingDetails[] | undefined; + const reportFindings: ReviewFinding[] | undefined = reportFindingDetails?.map(toReviewFinding); + const finalized = finalizeSubprocessOutput({ + rawOutput, + exitCode, + stderr, + doneAborted: Boolean(done.aborted), + signalAborted: Boolean(signal?.aborted), + yieldItems, + reportFindings, + outputSchema: args.outputSchema, + }); + rawOutput = finalized.rawOutput; + exitCode = finalized.exitCode; + stderr = finalized.stderr; + // Salvage for cancelled/aborted children that produced no completed output: + // surface the last assistant text + stats instead of "(no output)" so the + // parent doesn't redo work the child already finished. + const salvageText = monitor.lastAssistantSalvageText(); + if ( + (done.aborted || signal?.aborted || monitor.runtimeLimitExceeded()) && + !rawOutput.trim() && + salvageText !== undefined + ) { + rawOutput = `[cancelled after ${progress.requests} req, ${progress.tokens} tok — last activity: "${formatSalvageSnippet(salvageText)}"]`; + } + const lastYield = yieldItems?.[yieldItems.length - 1]; + const yieldAbortReason = lastYield?.status === "aborted" ? lastYield.error || "Subagent aborted task" : undefined; + const { abortedViaYield, hasYield } = finalized; + const { content: truncatedOutput, truncated } = truncateTail(rawOutput, { + maxBytes: MAX_OUTPUT_BYTES, + maxLines: MAX_OUTPUT_LINES, + }); + + // Write output artifact (input and jsonl already written in real-time) + // Compute output metadata for agent:// URL integration + let outputMeta: { lineCount: number; charCount: number } | undefined; + let outputPath: string | undefined; + if (args.artifactsDir) { + outputPath = path.join(args.artifactsDir, `${id}.md`); + try { + await Bun.write(outputPath, rawOutput); + outputMeta = { + lineCount: rawOutput.split("\n").length, + charCount: rawOutput.length, + }; + } catch { + // Non-fatal + } + } + + // Update final progress. A wall-clock timeout always wins: if the runtime + // limit fired we report aborted/failed regardless of whether a yield landed + // while we were tearing the session down. The yield data is still surfaced + // to the caller via `progress.extractedToolData`, but the exit status must + // reflect the timeout so on-call doesn't mistake a stuck run for success. + const runtimeLimitExceeded = monitor.runtimeLimitExceeded(); + if (runtimeLimitExceeded && exitCode === 0) { + exitCode = 1; + } + const wasAborted = + runtimeLimitExceeded || abortedViaYield || (!hasYield && (done.aborted || signal?.aborted || false)); + const finalAbortReason = wasAborted + ? runtimeLimitExceeded + ? monitor.resolveAbortReasonText() + : abortedViaYield + ? yieldAbortReason + : (done.abortReason ?? + (signal?.aborted ? monitor.resolveSignalAbortReason() : monitor.resolveAbortReasonText())) + : undefined; + progress.status = wasAborted ? "aborted" : exitCode === 0 ? "completed" : "failed"; + monitor.scheduleProgress(true); + + // Emit lifecycle end event after finalization so yield status is reflected + if (args.eventBus) { + args.eventBus.emit(TASK_SUBAGENT_LIFECYCLE_CHANNEL, { + id, + agent: agent.name, + parentToolCallId: args.parentToolCallId, + agentSource: agent.source, + description: args.description, + status: progress.status as "completed" | "failed" | "aborted", + sessionFile: args.sessionFile, + index, + }); + } + + return { + index, + id, + agent: agent.name, + agentSource: agent.source, + task, + assignment, + description: args.description, + lastIntent: progress.lastIntent, + exitCode, + output: truncatedOutput, + stderr, + truncated: Boolean(truncated), + durationMs: Date.now() - args.startTime, + tokens: progress.tokens, + requests: progress.requests, + contextTokens: progress.contextTokens, + contextWindow: progress.contextWindow, + modelOverride, + resolvedModel: progress.resolvedModel, + error: exitCode !== 0 && stderr ? stderr : undefined, + aborted: wasAborted, + abortReason: finalAbortReason, + usage: monitor.hasUsage() ? monitor.accumulatedUsage : undefined, + outputPath, + extractedToolData: progress.extractedToolData, + retryFailure: progress.retryFailure, + outputMeta, + }; +} + +/** + * Run a single agent in-process. + */ +export async function runSubprocess(options: ExecutorOptions): Promise { + const { + cwd, + agent, + task, + assignment, + index, + id, + worktree, + modelOverride, + thinkingLevel, + outputSchema, + enableLsp, + signal, + onProgress, + } = options; + const startTime = Date.now(); + + // Check if already aborted + if (signal?.aborted) { + return { + index, + id, + agent: agent.name, + agentSource: agent.source, + task, + assignment, + description: options.description, + exitCode: 1, + output: "", + stderr: "Cancelled before start", + truncated: false, + durationMs: 0, + tokens: 0, + requests: 0, + modelOverride, + error: "Cancelled before start", + aborted: true, + abortReason: "Cancelled before start", + }; + } + + // Set up artifact paths and write input file upfront if artifacts dir provided + let subtaskSessionFile: string | undefined; + if (options.artifactsDir) { + subtaskSessionFile = path.join(options.artifactsDir, `${id}.jsonl`); + } + + const settings = options.settings ?? Settings.isolated(); + const subagentSettings = createSubagentSettings( + settings, + agent.readSummarize === false ? { "read.summarize.enabled": false } : undefined, + ); + const maxRecursionDepth = settings.get("task.maxRecursionDepth") ?? 2; + const maxRuntimeMs = Math.max( + 0, + Math.trunc(Number(options.maxRuntimeMs ?? settings.get("task.maxRuntimeMs") ?? 0) || 0), + ); + // TTL before an adopted idle subagent is parked by the lifecycle manager. + // <= 0 disables parking (the session stays live until process teardown). + const agentIdleTtlMs = Math.trunc(Number(settings.get("task.agentIdleTtlMs") ?? 420_000) || 0); + const configuredDefaultBudget = Math.max( + 0, + Math.trunc(Number(settings.get("task.softRequestBudget") ?? SOFT_REQUEST_BUDGET.default) || 0), + ); + const softRequestBudget = + configuredDefaultBudget === 0 ? 0 : (SOFT_REQUEST_BUDGET[agent.name] ?? configuredDefaultBudget); + const parentDepth = options.taskDepth ?? 0; + const childDepth = parentDepth + 1; + const atMaxDepth = maxRecursionDepth >= 0 && childDepth >= maxRecursionDepth; + + // Add tools if specified + let toolNames: string[] | undefined; + if (agent.tools && agent.tools.length > 0) { + toolNames = agent.tools; + // Auto-include task tool if spawns defined but task not in tools + if (agent.spawns !== undefined && !toolNames.includes("task") && !atMaxDepth) { + toolNames = [...toolNames, "task"]; + } + } + + if (atMaxDepth && toolNames?.includes("task")) { + toolNames = toolNames.filter(name => name !== "task"); + } + // IRC is always available; the COOP prompt section advertises it, so a restricted + // whitelist must still carry `irc` for the subagent to actually use it. + if (toolNames && !toolNames.includes("irc")) { + toolNames = [...toolNames, "irc"]; + } + if (toolNames?.includes("exec")) { + const allowEvalPy = settings.get("eval.py") ?? true; + const allowEvalJs = settings.get("eval.js") ?? true; + const expanded = toolNames.filter(name => name !== "exec"); + if (allowEvalPy || allowEvalJs) expanded.push("eval"); + expanded.push("bash"); + toolNames = Array.from(new Set(expanded)); + } + + const modelPatterns = normalizeModelPatterns(modelOverride ?? agent.model); + const sessionFile = subtaskSessionFile ?? null; + const spawnsEnv = atMaxDepth + ? "" + : agent.spawns === undefined + ? "" + : agent.spawns === "*" + ? "*" + : agent.spawns.join(","); + + const lspEnabled = enableLsp ?? true; + const ircEnabled = subagentSettings.get("irc.enabled") === true; + const skipPythonPreflight = Array.isArray(toolNames) && !toolNames.includes("eval"); + + const monitor = createSubagentRunMonitor({ + index, + id, + agent, + task, + assignment, + description: options.description, + modelOverride, + signal, + onProgress, + eventBus: options.eventBus, + parentToolCallId: options.parentToolCallId, + sessionFile: subtaskSessionFile, + softRequestBudget, + maxRuntimeMs, + }); + const progress = monitor.progress; + let unsubscribe: (() => void) | null = null; + let reviveSession: (() => Promise) | null = null; + // Adopted (kept-alive) subagents flip registry status from session events on + // later turns: revive/wake → running, turn drained → idle. The subscription + // intentionally survives this run; a disposed session emits nothing, so it + // needs no teardown. + const installRegistryStatusSync = (target: AgentSession): void => { + target.subscribe(event => { + if (event.type === "agent_start") { + AgentRegistry.global().setStatus(id, "running"); + } else if (event.type === "agent_end") { + AgentRegistry.global().setStatus(id, "idle"); + } + }); + }; + const runSubagent = async (): Promise<{ exitCode: number; error?: string; @@ -1155,17 +1703,13 @@ export async function runSubprocess(options: ExecutorOptions): Promise => { const sessionAbortController = new AbortController(); + const abortSignal = monitor.abortSignal; let exitCode = 0; let error: string | undefined; let aborted = false; let abortReasonText: string | undefined; const checkAbort = () => { if (abortSignal.aborted) { - aborted = abortReason === "signal" || runtimeLimitExceeded || abortReason === undefined; - if (aborted) { - abortReasonText ??= resolveAbortReasonText(); - } - exitCode = 1; throw new ToolAbortError(); } }; @@ -1283,7 +1827,11 @@ export async function runSubprocess(options: ExecutorOptions): Promise ({ cwd: worktree ?? cwd, authStorage, modelRegistry, @@ -1303,12 +1851,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise { const subagentPrompt = prompt.render(subagentSystemPromptTemplate, { agent: agent.systemPrompt, - context: options.context?.trim() ?? "", planReference: options.planReference?.content ?? "", planReferencePath: options.planReference?.path ?? "", worktree: worktree ?? "", outputSchema: normalizedOutputSchema, - contextFile: contextFileForPrompt, ircPeers: ircEnabled ? renderIrcPeerRoster(id) : "", ircSelfId: ircEnabled ? id : "", }); @@ -1316,7 +1862,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { + const reopened = await SessionManager.open(sessionFile); + if (options.parentArtifactManager) { + reopened.adoptArtifactManager(options.parentArtifactManager); + } + const { session: revived } = await createAgentSession(buildSubagentSessionOptions(reopened)); + installRegistryStatusSync(revived); + return revived; + }; + } // Emit lifecycle start event if (options.eventBus) { @@ -1449,44 +2013,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { - emitSubagentEvent(event); - if (event.type === "auto_retry_start") { - progress.retryState = { - attempt: event.attempt, - maxAttempts: event.maxAttempts, - delayMs: event.delayMs, - errorMessage: event.errorMessage, - startedAtMs: Date.now(), - }; - progress.retryFailure = undefined; - scheduleProgress(true); - return; - } - if (event.type === "auto_retry_end") { - const attempt = progress.retryState?.attempt ?? event.attempt; - progress.retryState = undefined; - if (!event.success) { - progress.retryFailure = { - attempt, - errorMessage: event.finalError ?? "Auto-retry failed", - }; - } - scheduleProgress(true); - return; - } - if (isAgentEvent(event)) { - try { - processEvent(event); - } catch (err) { - logger.error("Subagent event processing failed", { - error: err instanceof Error ? err.message : String(err), - }); - requestAbort("terminate"); - } - } - }); + unsubscribe = monitor.attach(session); checkAbort(); // Autoload skills via sendCustomMessage (same mechanic as /skill:) @@ -1504,78 +2031,12 @@ export async function runSubprocess(options: ExecutorOptions): Promise= MAX_YIELD_RETRIES; - await awaitAbortable( - session.prompt(reminder, { - attribution: "agent", - synthetic: true, - ...(isFinalRetry && reminderToolChoice ? { toolChoice: reminderToolChoice } : {}), - }), - ); - await awaitAbortable(session.waitForIdle()); - } catch (err) { - if (abortSignal.aborted || err instanceof ToolAbortError) { - // Benign control-flow exit — user cancel (^C) or compaction aborting - // pending operations both surface here as ToolAbortError. The outer - // catch and finally already mark the run aborted; logging at ERROR - // would spam operator dashboards with non-failures. - logger.debug("Subagent prompt aborted", { - reason: abortReason ?? "signal", - }); - } else { - logger.error("Subagent prompt failed", { - error: err instanceof Error ? err.message : String(err), - }); - } - } - } - - await awaitAbortable(session.waitForIdle()); - if (!yieldCalled && !abortSignal.aborted) { - exitCode = 0; - } - - const lastAssistant = session.getLastAssistantMessage(); - if (lastAssistant) { - if (lastAssistant.stopReason === "aborted") { - aborted = abortReason === "signal" || runtimeLimitExceeded || abortReason === undefined; - if (aborted) { - // A real caller signal or the wall-clock timer carries a precise - // reason (signal.reason / "runtime limit exceeded"). An internal - // turn abort (abortReason === undefined) does NOT — prefer the - // assistant message's own errorMessage ("Request was aborted" or a - // specific stream error) over the misleading "Cancelled by caller". - abortReasonText ??= - abortReason === "signal" || runtimeLimitExceeded - ? resolveAbortReasonText() - : lastAssistant.errorMessage?.trim() || resolveAbortReasonText(); - } - exitCode = 1; - } else if (lastAssistant.stopReason === "error") { - exitCode = 1; - error ??= lastAssistant.errorMessage || "Subagent failed"; - } - } + const outcome = await driveSessionToYield(session, monitor, task); + exitCode = outcome.exitCode; + error = outcome.error; + aborted = outcome.aborted; + abortReasonText = outcome.abortReasonText; } catch (err) { exitCode = 1; if (!abortSignal.aborted) { @@ -1583,9 +2044,9 @@ export async function runSubprocess(options: ExecutorOptions): Promise session.dispose()); - } catch { - // Ignore cleanup errors + const session = monitor.takeActiveSession(); + if (session) { + monitor.captureSalvage(session); + const registry = AgentRegistry.global(); + if (aborted) { + // Hard abort (caller signal / wall-clock / budget): terminal teardown. + registry.setStatus(id, "aborted"); + try { + await untilAborted(AbortSignal.timeout(5000), () => session.dispose()); + } catch { + // Ignore cleanup errors + } + } else if (worktree !== undefined) { + // Isolated run: the worktree is merged + cleaned after the run, so + // the session is not resumable. Park the ref WITHOUT adopting — the + // transcript stays reachable (history://), but ensureLive will throw. + // Status must flip to "parked" before dispose so the sdk dispose + // wrapper skips unregister. + registry.setStatus(id, "parked"); + try { + await untilAborted(AbortSignal.timeout(5000), () => session.dispose()); + } catch { + // Ignore cleanup errors + } + registry.detachSession(id); + } else { + // Keep-alive: finished and failed subagents both stay interrogable. + // The lifecycle manager owns idle-TTL parking + revival from here on. + registry.setStatus(id, "idle"); + AgentLifecycleManager.global().adopt(id, { + idleTtlMs: agentIdleTtlMs, + revive: reviveSession ?? undefined, + }); } } } @@ -1619,87 +2106,115 @@ export async function runSubprocess(options: ExecutorOptions): Promise 0 ? finalOutputChunks.join("") : outputChunks.join(""); - const yieldItems = progress.extractedToolData?.yield as YieldItem[] | undefined; - const reportFindingDetails = progress.extractedToolData?.report_finding as ReportFindingDetails[] | undefined; - const reportFindings: ReviewFinding[] | undefined = reportFindingDetails?.map(toReviewFinding); - const finalized = finalizeSubprocessOutput({ - rawOutput, - exitCode, - stderr, - doneAborted: Boolean(done.aborted), - signalAborted: Boolean(signal?.aborted), - yieldItems, - reportFindings, + return finalizeRunResult({ + monitor, + done, + index, + id, + agent, + task, + assignment, + description: options.description, + modelOverride, outputSchema, + signal, + artifactsDir: options.artifactsDir, + eventBus: options.eventBus, + parentToolCallId: options.parentToolCallId, + sessionFile: subtaskSessionFile, + startTime, }); - rawOutput = finalized.rawOutput; - exitCode = finalized.exitCode; - stderr = finalized.stderr; - const lastYield = yieldItems?.[yieldItems.length - 1]; - const yieldAbortReason = lastYield?.status === "aborted" ? lastYield.error || "Subagent aborted task" : undefined; - const { abortedViaYield, hasYield } = finalized; - const { content: truncatedOutput, truncated } = truncateTail(rawOutput, { - maxBytes: MAX_OUTPUT_BYTES, - maxLines: MAX_OUTPUT_LINES, - }); +} - // Write output artifact (input and jsonl already written in real-time) - // Compute output metadata for agent:// URL integration - let outputMeta: { lineCount: number; charCount: number } | undefined; - let outputPath: string | undefined; - if (options.artifactsDir) { - outputPath = path.join(options.artifactsDir, `${id}.md`); - try { - await Bun.write(outputPath, rawOutput); - outputMeta = { - lineCount: rawOutput.split("\n").length, - charCount: rawOutput.length, - }; - } catch { - // Non-fatal - } +/** Options for resuming an existing live subagent session with a follow-up assignment. */ +export interface ResumeExecutorOptions { + /** Live session, e.g. from `AgentLifecycleManager.global().ensureLive(id)`. */ + session: AgentSession; + /** Registry agent id being resumed. */ + id: string; + /** Agent definition for progress labels and soft budgets; a minimal stub is acceptable. */ + agent: AgentDefinition; + /** Rendered follow-up prompt, injected via the session's normal prompt path. */ + task: string; + assignment?: string; + description?: string; + index: number; + parentToolCallId?: string; + /** Optional schema validating this follow-up's yield payload. */ + outputSchema?: unknown; + signal?: AbortSignal; + onProgress?: (progress: AgentProgress) => void; + eventBus?: EventBus; + settings?: Settings; + /** Where the `.md` output artifact is (over)written for this assignment. */ + artifactsDir?: string; +} + +/** + * Run a follow-up assignment on an EXISTING live agent session through the same + * monitoring/finalize pipeline as a fresh spawn. The session is never created + * or disposed here: it stays alive (and adopted by the lifecycle manager from + * its original spawn) afterwards — registry status flips via the session's + * registry status sync, and the idle TTL re-arms via the lifecycle manager's + * registry subscription. Each resume overwrites the `agent://` output + * artifact; the transcript accretes in the session JSONL. + */ +export async function resumeSubprocess(options: ResumeExecutorOptions): Promise { + const { session, id, agent, task, assignment, index, signal } = options; + const startTime = Date.now(); + + if (signal?.aborted) { + return { + index, + id, + agent: agent.name, + agentSource: agent.source, + task, + assignment, + description: options.description, + exitCode: 1, + output: "", + stderr: "Cancelled before start", + truncated: false, + durationMs: 0, + tokens: 0, + requests: 0, + error: "Cancelled before start", + aborted: true, + abortReason: "Cancelled before start", + }; } - // Update final progress. A wall-clock timeout always wins: if the runtime - // limit fired we report aborted/failed regardless of whether a yield landed - // while we were tearing the session down. The yield data is still surfaced - // to the caller via `progress.extractedToolData`, but the exit status must - // reflect the timeout so on-call doesn't mistake a stuck run for success. - if (runtimeLimitExceeded && exitCode === 0) { - exitCode = 1; - } - const wasAborted = - runtimeLimitExceeded || abortedViaYield || (!hasYield && (done.aborted || signal?.aborted || false)); - const finalAbortReason = wasAborted - ? runtimeLimitExceeded - ? resolveAbortReasonText() - : abortedViaYield - ? yieldAbortReason - : (done.abortReason ?? (signal?.aborted ? resolveSignalAbortReason() : resolveAbortReasonText())) - : undefined; - progress.status = wasAborted ? "aborted" : exitCode === 0 ? "completed" : "failed"; - scheduleProgress(true); + const settings = options.settings ?? Settings.isolated(); + const maxRuntimeMs = Math.max(0, Math.trunc(Number(settings.get("task.maxRuntimeMs") ?? 0) || 0)); + const configuredDefaultBudget = Math.max( + 0, + Math.trunc(Number(settings.get("task.softRequestBudget") ?? SOFT_REQUEST_BUDGET.default) || 0), + ); + const softRequestBudget = + configuredDefaultBudget === 0 ? 0 : (SOFT_REQUEST_BUDGET[agent.name] ?? configuredDefaultBudget); + const sessionFile = AgentRegistry.global().get(id)?.sessionFile ?? undefined; + + const monitor = createSubagentRunMonitor({ + index, + id, + agent, + task, + assignment, + description: options.description, + signal, + onProgress: options.onProgress, + eventBus: options.eventBus, + parentToolCallId: options.parentToolCallId, + sessionFile, + softRequestBudget, + maxRuntimeMs, + }); + monitor.setActiveSession(session); + const unsubscribe = monitor.attach(session); - // Emit lifecycle end event after finalization so yield status is reflected if (options.eventBus) { options.eventBus.emit(TASK_SUBAGENT_LIFECYCLE_CHANNEL, { id, @@ -1707,38 +2222,56 @@ export async function runSubprocess(options: ExecutorOptions): Promise): void { - const input = usage.input ?? 0; - const output = usage.output ?? 0; - const cacheRead = usage.cacheRead ?? 0; - const cacheWrite = usage.cacheWrite ?? 0; - const totalTokens = usage.totalTokens ?? input + output + cacheRead + cacheWrite; - const cost = - usage.cost ?? - ({ - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - total: 0, - } satisfies Usage["cost"]); - - target.input += input; - target.output += output; - target.cacheRead += cacheRead; - target.cacheWrite += cacheWrite; - target.totalTokens += totalTokens; - target.cost.input += cost.input; - target.cost.output += cost.output; - target.cost.cacheRead += cost.cacheRead; - target.cost.cacheWrite += cost.cacheWrite; - target.cost.total += cost.total; -} // Re-export types and utilities export { loadBundledAgents as BUNDLED_AGENTS } from "./agents"; @@ -165,6 +132,17 @@ export function isReadOnlyAgent(agent: AgentDefinition): boolean { return !!agent.tools?.length && agent.tools.every(tool => READ_ONLY_TOOL_NAMES.has(tool)); } +/** + * Preview text for a child result. Falls back to "(no output)" — annotated + * with the request count when the child actually did work, so the parent can + * tell a no-op child from one that burned requests before being cancelled. + */ +export function formatResultOutputFallback(result: Pick): string { + const base = result.output.trim() || result.stderr.trim(); + if (base) return base; + return result.requests > 0 ? `(no output) after ${result.requests} req` : "(no output)"; +} + /** * Render the tool description from a cached agent list and current settings. */ @@ -172,7 +150,6 @@ function renderDescription( agents: AgentDefinition[], maxConcurrency: number, isolationEnabled: boolean, - asyncEnabled: boolean, disabledAgents: string[], simpleMode: TaskSimpleMode, ircEnabled: boolean, @@ -196,14 +173,12 @@ function renderDescription( description: agent.description, readOnly: isReadOnlyAgent(agent), })); - const { contextEnabled, customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); + const { customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); return prompt.render(taskDescriptionTemplate, { agents: renderedAgents, spawningDisabled, MAX_CONCURRENCY: maxConcurrency, isolationEnabled, - asyncEnabled, - contextEnabled, customSchemaEnabled, ircEnabled, defaultMode: simpleMode === "default", @@ -220,87 +195,46 @@ function createTaskModeError(text: string): AgentToolResult { } function validateTaskModeParams(simpleMode: TaskSimpleMode, params: TaskParams): string | undefined { - const { contextEnabled, customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); - const disallowedFields: string[] = []; - if (!contextEnabled && params.context !== undefined) { - disallowedFields.push("context"); - } - if (!customSchemaEnabled && params.schema !== undefined) { - disallowedFields.push("schema"); - } - if (disallowedFields.length === 0) { + const { customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); + if (customSchemaEnabled || params.schema === undefined) { return undefined; } - - if (simpleMode === "schema-free") { - return "task.simple is set to schema-free, so the task tool does not accept `schema`. Remove it and rely on the selected agent definition or inherited session schema."; - } - - if (disallowedFields.length === 1) { - return `task.simple is set to independent, so the task tool does not accept \`${disallowedFields[0]}\`. Put everything the subagent needs inside each task assignment.`; - } - - return "task.simple is set to independent, so the task tool does not accept `context` or `schema`. Put all required background and output expectations inside each task assignment or the selected agent definition."; + return `task.simple is set to ${simpleMode}, so the task tool does not accept \`schema\`. Remove it and rely on the selected agent definition or inherited session schema.`; } -/** Sentinel for async jobs whose subagent finished with a failing result; batch counters are already updated. */ -class TaskJobError extends Error {} - /** - * Validate task ids: every task needs a non-empty id and ids must be unique - * (case-insensitive). Returns a problem description, or undefined when valid. + * Validate the spawn/resume parameter contract: `agent` XOR `resume`, + * `resume` excludes `isolated`, and `assignment` is always required. + * Returns a problem description, or undefined when valid. */ -function validateTaskIds(tasks: TaskParams["tasks"]): string | undefined { - const missingTaskIndexes: number[] = []; - const idIndexes = new Map(); - - for (let i = 0; i < tasks.length; i++) { - const id = tasks[i]?.id; - if (typeof id !== "string" || id.trim() === "") { - missingTaskIndexes.push(i); - continue; - } - const normalizedId = id.toLowerCase(); - const indexes = idIndexes.get(normalizedId); - if (indexes) { - indexes.push(i); - } else { - idIndexes.set(normalizedId, [i]); - } +function validateSpawnParams(params: TaskParams): string | undefined { + const resume = typeof params.resume === "string" ? params.resume.trim() : ""; + const agent = typeof params.agent === "string" ? params.agent.trim() : ""; + if (resume && agent) { + return "Provide either `agent` (spawn a new subagent) or `resume` (continue an existing one), not both."; } - - const duplicateIds: Array<{ id: string; indexes: number[] }> = []; - for (const [normalizedId, indexes] of idIndexes.entries()) { - if (indexes.length > 1) { - duplicateIds.push({ - id: tasks[indexes[0]]?.id ?? normalizedId, - indexes, - }); - } + if (!resume && !agent) { + return "Missing `agent`. Provide `agent` to spawn a subagent, or `resume` with an existing agent id."; } - - if (missingTaskIndexes.length === 0 && duplicateIds.length === 0) { - return undefined; + if (resume && params.isolated === true) { + return "`resume` cannot be combined with `isolated` — isolated agents are not resumable."; } - - const problems: string[] = []; - if (missingTaskIndexes.length > 0) { - problems.push(`Missing task ids at indexes: ${missingTaskIndexes.join(", ")}`); + if (typeof params.assignment !== "string" || params.assignment.trim() === "") { + return "Missing `assignment`. Provide complete, self-contained instructions for the agent."; } - if (duplicateIds.length > 0) { - const details = duplicateIds.map(entry => `${entry.id} (indexes ${entry.indexes.join(", ")})`).join("; "); - problems.push(`Duplicate task ids detected (case-insensitive): ${details}`); - } - return `Invalid tasks: ${problems.join(". ")}`; + return undefined; } +/** Sentinel for async jobs whose subagent finished with a failing result; progress is already updated. */ +class TaskJobError extends Error {} + /** * Process-level memo for create-time agent discovery, keyed by resolved cwd. * * `TaskTool.create` runs for every (sub)agent session in this process and the * walk-up + plugin-registry scan in `discoverAgents` is identical for a given * cwd, so repeat creations reuse the first scan. Execution-time discovery - * (`#executeSync`) intentionally stays fresh. The memo also tracks the live + * (`#runSpawn`) intentionally stays fresh. The memo also tracks the live * `discoverAgents` binding: test spies swap that binding, which invalidates * the memo automatically. */ @@ -332,8 +266,9 @@ function discoverAgentsForCreate(cwd: string): Promise { /** * Task tool - Delegate tasks to specialized agents. * - * Requires async initialization to discover available agents. - * Use `TaskTool.create(session)` to instantiate. + * Each call spawns ONE subagent (or resumes an existing one). Spawning is + * non-blocking: the call registers an AsyncJobManager job and returns + * immediately; the result is delivered when the agent yields. */ export class TaskTool implements AgentTool { readonly name = "task"; @@ -341,22 +276,21 @@ export class TaskTool implements AgentTool { const params = args as Partial; const lines: string[] = []; - if (typeof params.agent === "string") { + if (typeof params.resume === "string" && params.resume.trim()) { + lines.push(`Resume: ${truncateForPrompt(params.resume)}`); + } else if (typeof params.agent === "string") { lines.push(`Agent: ${truncateForPrompt(params.agent)}`); } - const tasks = Array.isArray(params.tasks) ? params.tasks : []; - const firstTask = tasks[0]; - if (firstTask) { - lines.push(`Task: ${truncateForPrompt(firstTask.id)}`); - lines.push(`Assignment:\n${truncateForPrompt(firstTask.assignment)}`); - if (tasks.length > 1) { - lines.push(`+${tasks.length - 1} more task${tasks.length === 2 ? "" : "s"}`); - } + if (typeof params.id === "string" && params.id.trim()) { + lines.push(`Task: ${truncateForPrompt(params.id)}`); + } + if (typeof params.assignment === "string") { + lines.push(`Assignment:\n${truncateForPrompt(params.assignment)}`); } return lines; }; readonly label = "Task"; - readonly summary = "Spawn a subagent to complete a parallel task"; + readonly summary = "Spawn a subagent to complete a task in the background"; readonly strict = true; readonly loadMode = "discoverable"; readonly renderResult = renderResult; @@ -366,6 +300,12 @@ export class TaskTool implements AgentTool> { const params = repairTaskParams(rawParams as TaskParams); const simpleMode = this.#getTaskSimpleMode(); - const validationError = validateTaskModeParams(simpleMode, params); + const validationError = validateTaskModeParams(simpleMode, params) ?? validateSpawnParams(params); if (validationError) { return createTaskModeError(validationError); } - const asyncEnabled = this.session.settings.get("async.enabled"); - const selectedAgent = this.#discoveredAgents.find(agent => agent.name === params.agent); - if (!asyncEnabled || selectedAgent?.blocking === true) { - return this.#executeSync(toolCallId, params, signal, onUpdate); - } - + const isResume = typeof params.resume === "string" && params.resume.trim().length > 0; + const selectedAgent = isResume ? undefined : this.#discoveredAgents.find(agent => agent.name === params.agent); const manager = this.session.asyncJobManager; - if (!manager) { - // Async was requested but no manager is registered (e.g. an - // orphaned session whose host never wired one up). Falling back - // to the sync path keeps the tool usable; only background/job-poll - // semantics are lost. - logger.warn("task: async.enabled but no AsyncJobManager registered; falling back to sync execution"); - return this.#executeSync(toolCallId, params, signal, onUpdate); + if (!manager || selectedAgent?.blocking === true) { + // Sync fallback: orphaned host that never wired a job manager, or an + // agent definition that declares `blocking: true`. The session-scoped + // semaphore still bounds fan-out across parallel task calls. + if (!manager) { + logger.warn("task: no AsyncJobManager registered; falling back to sync execution"); + } + const semaphore = this.#getSpawnSemaphore(); + await semaphore.acquire(); + try { + return await this.#executeSync(toolCallId, params, signal, onUpdate); + } finally { + semaphore.release(); + } } - const taskItems = params.tasks ?? []; - if (taskItems.length === 0) { - return this.#executeSync(toolCallId, params, signal, onUpdate); + // Resolve the agent id up front so the immediate result can name it. + let agentId: string; + if (isResume) { + agentId = params.resume!.trim(); + if (!AgentRegistry.global().get(agentId)) { + throw new ToolError( + `Unknown agent "${agentId}" — nothing to resume. Use \`irc\` op:"list" to see live agent ids; past transcripts are readable at history://${agentId}.`, + ); + } + } else { + const outputManager = + this.session.agentOutputManager ?? new AgentOutputManager(this.session.getArtifactsDir ?? (() => null)); + agentId = await outputManager.allocate(params.id?.trim() || generateTaskName()); } - const taskIdProblem = validateTaskIds(taskItems); - if (taskIdProblem) { - return createTaskModeError(taskIdProblem); - } - - const outputManager = - this.session.agentOutputManager ?? new AgentOutputManager(this.session.getArtifactsDir ?? (() => null)); - const uniqueIds = await outputManager.allocateBatch(taskItems.map(t => t.id)); - const fallbackAgentSource = - this.#discoveredAgents.find(agent => agent.name === params.agent)?.source ?? "bundled"; - const progressByTaskId = new Map(); - for (let index = 0; index < taskItems.length; index++) { - const taskItem = taskItems[index]; - const assignment = taskItem.assignment.trim(); - progressByTaskId.set(taskItem.id, { - index, - id: taskItem.id, - agent: params.agent, - agentSource: fallbackAgentSource, - status: "pending", - task: renderSubagentUserPrompt(assignment, simpleMode), - assignment, - description: taskItem.description, - recentTools: [], - recentOutput: [], - toolCount: 0, - tokens: 0, - cost: 0, - durationMs: 0, - }); - } - - const startedJobs: Array<{ jobId: string; taskId: string }> = []; - const failedSchedules: string[] = []; - let completedJobs = 0; - let failedJobs = 0; - - const getProgressSnapshot = (): AgentProgress[] => { - // Shallow copies: top-level fields are reassigned (never mutated in - // place) and the large nested payloads (extractedToolData) are - // immutable once attached — structuredClone here cost O(batch × payload) - // per progress event. - return Array.from(progressByTaskId.values()) - .sort((a, b) => a.index - b.index) - .map(progress => ({ ...progress })); + const assignment = (params.assignment ?? "").trim(); + const agentLabel = isResume + ? (AgentRegistry.global().get(agentId)?.displayName ?? "task") + : (params.agent ?? "task"); + const progress: AgentProgress = { + index: 0, + id: agentId, + agent: agentLabel, + agentSource: selectedAgent?.source ?? "bundled", + status: "pending", + task: renderSubagentUserPrompt(assignment, simpleMode), + assignment, + description: params.description, + recentTools: [], + recentOutput: [], + toolCount: 0, + requests: 0, + tokens: 0, + cost: 0, + durationMs: 0, }; const buildAsyncDetails = (state: "running" | "completed" | "failed", jobId: string): TaskToolDetails => ({ projectAgentsDir: null, results: [], totalDurationMs: 0, - progress: getProgressSnapshot(), + progress: [{ ...progress }], async: { state, jobId, type: "task" }, }); - const emitAsyncUpdate = (state: "running" | "completed" | "failed", text: string): void => { - const primaryJobId = startedJobs[0]?.jobId ?? "task"; - onUpdate?.({ - content: [{ type: "text", text }], - details: buildAsyncDetails(state, primaryJobId), - }); + const buildResumeHint = (aborted: boolean): string => { + if (aborted) { + return `\n\n${agentId} was aborted — transcript at history://${agentId}`; + } + return `\n\n${agentId} is now idle — task(resume:"${agentId}") to continue it, transcript at history://${agentId}`; }; - const maxConcurrency = this.session.settings.get("task.maxConcurrency"); - const semaphore = new Semaphore(maxConcurrency); - - for (let i = 0; i < taskItems.length; i++) { - const taskItem = taskItems[i]; - if (signal?.aborted) { - failedSchedules.push(`${taskItem.id}: cancelled before scheduling`); - completedJobs += 1; - const progress = progressByTaskId.get(taskItem.id); - if (progress) { - progress.status = "aborted"; - } - continue; - } - - const uniqueId = uniqueIds[i]; - const singleParams: TaskParams = { ...params, tasks: [taskItem] }; - const label = uniqueId; - try { - const jobId = manager.register( - "task", - label, - async ({ signal: runSignal, reportProgress, markRunning }) => { - const startedAt = Date.now(); - const progress = progressByTaskId.get(taskItem.id); - await semaphore.acquire(); - if (runSignal.aborted) { - semaphore.release(); - if (progress) { - progress.status = "aborted"; - } - completedJobs += 1; - failedJobs += 1; - throw new Error("Aborted before execution"); - } - markRunning(); - if (progress) { - progress.status = "running"; - } + let jobId: string; + try { + jobId = manager.register( + "task", + agentId, + async ({ jobId: ownJobId, signal: runSignal, reportProgress, markRunning }) => { + const startedAt = Date.now(); + const semaphore = this.#getSpawnSemaphore(); + await semaphore.acquire(); + if (runSignal.aborted) { + semaphore.release(); + progress.status = "aborted"; + throw new Error("Aborted before execution"); + } + markRunning(); + progress.status = "running"; + await reportProgress( + `Running background task ${agentId}...`, + buildAsyncDetails("running", ownJobId) as unknown as Record, + ); + try { + const result = await this.#executeSync(toolCallId, params, runSignal, undefined, agentId); + const finalText = result.content.find(part => part.type === "text")?.text ?? "(no output)"; + const singleResult = result.details?.results[0]; + // A missing result means the sync path failed at the tool level + // (results: []) — treat it as a failure, not success. + const resultFailed = !singleResult || (singleResult.aborted ?? false) || singleResult.exitCode !== 0; + progress.status = singleResult?.aborted ? "aborted" : resultFailed ? "failed" : "completed"; + progress.durationMs = singleResult?.durationMs ?? Math.max(0, Date.now() - startedAt); + progress.tokens = singleResult?.tokens ?? 0; + progress.requests = singleResult?.requests ?? 0; + progress.contextTokens = singleResult?.contextTokens; + progress.contextWindow = singleResult?.contextWindow; + progress.cost = singleResult?.usage?.cost.total ?? 0; + progress.extractedToolData = singleResult?.extractedToolData; + progress.retryFailure = singleResult?.retryFailure; + progress.retryState = undefined; + const statusText = resultFailed + ? `Background task ${agentId} failed.` + : `Background task ${agentId} complete.`; await reportProgress( - `Running background task ${taskItem.id}...`, - buildAsyncDetails("running", startedJobs[0]?.jobId ?? label) as unknown as Record, + statusText, + buildAsyncDetails(resultFailed ? "failed" : "completed", ownJobId) as unknown as Record< + string, + unknown + >, ); - try { - const result = await this.#executeSync(toolCallId, singleParams, runSignal, undefined, [uniqueId]); - const finalText = result.content.find(part => part.type === "text")?.text ?? "(no output)"; - const singleResult = result.details?.results[0]; - // A missing per-task result means #executeSync failed at the - // tool level (results: []) — treat it as a failure, not success. - const resultFailed = - !singleResult || (singleResult.aborted ?? false) || singleResult.exitCode !== 0; - if (progress) { - progress.status = singleResult?.aborted ? "aborted" : resultFailed ? "failed" : "completed"; - progress.durationMs = singleResult?.durationMs ?? Math.max(0, Date.now() - startedAt); - progress.tokens = singleResult?.tokens ?? 0; - progress.contextTokens = singleResult?.contextTokens; - progress.contextWindow = singleResult?.contextWindow; - progress.cost = singleResult?.usage?.cost.total ?? 0; - progress.extractedToolData = singleResult?.extractedToolData; - progress.retryFailure = singleResult?.retryFailure; - progress.retryState = undefined; - } - completedJobs += 1; - if (resultFailed) { - failedJobs += 1; - } - const remaining = taskItems.length - completedJobs; - const isDone = remaining === 0; - await reportProgress( - isDone - ? `Background task batch complete: ${completedJobs}/${taskItems.length} finished.` - : `Background task batch progress: ${completedJobs}/${taskItems.length} finished (${remaining} running).`, - buildAsyncDetails( - isDone ? (failedJobs > 0 || failedSchedules.length > 0 ? "failed" : "completed") : "running", - startedJobs[0]?.jobId ?? label, - ) as unknown as Record, - ); - if (isDone) { - emitAsyncUpdate( - failedJobs > 0 || failedSchedules.length > 0 ? "failed" : "completed", - `Background task batch complete: ${completedJobs}/${taskItems.length} finished.`, - ); - } - if (resultFailed) { - // Mark the job itself failed; counters above are already updated. - throw new TaskJobError(finalText); - } - return finalText; - } catch (error) { - if (error instanceof TaskJobError) { - throw error; - } - if (progress) { - progress.status = "failed"; - progress.durationMs = Math.max(0, Date.now() - startedAt); - } - completedJobs += 1; - failedJobs += 1; - const remaining = taskItems.length - completedJobs; - const isDone = remaining === 0; - await reportProgress( - isDone - ? `Background task batch complete with failures: ${failedJobs} failed.` - : `Background task batch progress: ${completedJobs}/${taskItems.length} finished (${remaining} running).`, - buildAsyncDetails( - isDone ? "failed" : "running", - startedJobs[0]?.jobId ?? label, - ) as unknown as Record, - ); - if (isDone) { - emitAsyncUpdate( - "failed", - `Background task batch complete with failures: ${failedJobs} failed.`, - ); - } - throw error; - } finally { - semaphore.release(); + onUpdate?.({ + content: [{ type: "text", text: statusText }], + details: buildAsyncDetails(resultFailed ? "failed" : "completed", ownJobId), + }); + const deliveryText = `${finalText}${buildResumeHint(singleResult?.aborted === true)}`; + if (resultFailed) { + // Mark the job itself failed; the failed agent stays interrogable. + throw new TaskJobError(deliveryText); } + return deliveryText; + } catch (error) { + if (error instanceof TaskJobError) { + throw error; + } + progress.status = "failed"; + progress.durationMs = Math.max(0, Date.now() - startedAt); + const statusText = `Background task ${agentId} failed.`; + await reportProgress( + statusText, + buildAsyncDetails("failed", ownJobId) as unknown as Record, + ); + onUpdate?.({ + content: [{ type: "text", text: statusText }], + details: buildAsyncDetails("failed", ownJobId), + }); + const message = error instanceof Error ? error.message : String(error); + const hint = AgentRegistry.global().get(agentId) ? buildResumeHint(false) : ""; + throw new TaskJobError(`${message}${hint}`); + } finally { + semaphore.release(); + } + }, + { + id: agentId, + queued: true, + ownerId: this.session.getAgentId?.() ?? undefined, + onProgress: (text, details) => { + const progressDetails = + (details as TaskToolDetails | undefined) ?? buildAsyncDetails("running", agentId); + onUpdate?.({ content: [{ type: "text", text }], details: progressDetails }); }, - { - id: label, - queued: true, - ownerId: this.session.getAgentId?.() ?? undefined, - onProgress: (text, details) => { - const progressDetails = - (details as TaskToolDetails | undefined) ?? - buildAsyncDetails("running", startedJobs[0]?.jobId ?? label); - onUpdate?.({ content: [{ type: "text", text }], details: progressDetails }); - }, - }, - ); - startedJobs.push({ jobId, taskId: taskItem.id }); - } catch (error) { - const message = error instanceof Error ? error.message : String(error); - failedSchedules.push(`${taskItem.id}: ${message}`); - completedJobs += 1; - const progress = progressByTaskId.get(taskItem.id); - if (progress) { - progress.status = "failed"; - } - } - } - - if (startedJobs.length === 0) { - const failureText = `Failed to start background task jobs: ${failedSchedules.join("; ")}`; + }, + ); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); return { - content: [{ type: "text", text: failureText }], + content: [{ type: "text", text: `Failed to start background task job: ${message}` }], details: { projectAgentsDir: null, results: [], totalDurationMs: 0 }, }; } - emitAsyncUpdate( - "running", - `Launching ${startedJobs.length} background ${startedJobs.length === 1 ? "task" : "tasks"}...`, - ); - - const scheduleFailureSummary = - failedSchedules.length > 0 - ? ` Failed to schedule ${failedSchedules.length} task${failedSchedules.length === 1 ? "" : "s"}.` - : ""; - const ircEnabled = this.session.settings.get("irc.enabled") === true; - const taskIdByItemId = new Map(); - for (let i = 0; i < taskItems.length; i++) { - taskIdByItemId.set(taskItems[i].id, uniqueIds[i]); - } - const startedListing = startedJobs - .map(({ taskId, jobId }) => { - const id = taskIdByItemId.get(taskId) ?? taskId; - const desc = progressByTaskId.get(taskId)?.description; - const prefix = `- \`${id}\` (job \`${jobId}\`)`; - return desc ? `${prefix} — ${desc}` : prefix; - }) - .join("\n"); const coordinationHint = ircEnabled - ? ` DM these ids via \`irc\` to coordinate while they run; reach for \`job\` only to inspect (\`list\`), wait (\`poll\`), or cancel a stuck task.` - : ` Use \`job\` to inspect (\`list\`), wait (\`poll\`), or cancel a stuck task by id.`; + ? `DM \`${agentId}\` via \`irc\` to coordinate while it runs; use \`job\` only to inspect (\`list\`), wait (\`poll\`), or cancel a stuck task.` + : `Use \`job\` to inspect (\`list\`), wait (\`poll\`), or cancel a stuck task.`; + const verb = isResume ? "Resumed" : "Spawned"; + const descriptionSuffix = params.description ? ` — ${params.description}` : ""; + + onUpdate?.({ + content: [{ type: "text", text: `${verb} agent \`${agentId}\`...` }], + details: buildAsyncDetails("running", jobId), + }); return { content: [ { type: "text", - text: `Started ${startedJobs.length} background task job${startedJobs.length === 1 ? "" : "s"} using ${params.agent}.${scheduleFailureSummary} Results will be delivered when complete.\n${startedListing}\n${coordinationHint}`, + text: `${verb} agent \`${agentId}\` (job \`${jobId}\`)${descriptionSuffix}. The result will be delivered when it yields. ${coordinationHint}`, }, ], details: { projectAgentsDir: null, results: [], totalDurationMs: 0, - progress: getProgressSnapshot(), - async: { state: "running", jobId: startedJobs[0].jobId, type: "task" }, + progress: [{ ...progress }], + async: { state: "running", jobId, type: "task" }, }, }; } + /** + * Synchronous execution of one spawn or resume. Used as the body of every + * async job and directly by the sync fallback (no job manager / blocking + * agent) and by in-process callers that need the result inline (e.g. the + * commit flow's analyze_files tool). + */ async #executeSync( toolCallId: string, params: TaskParams, signal?: AbortSignal, onUpdate?: AgentToolUpdateCallback, - preAllocatedIds?: string[], + preAllocatedId?: string, + ): Promise> { + if (typeof params.resume === "string" && params.resume.trim().length > 0) { + return this.#executeResume(toolCallId, params, signal, onUpdate); + } + return this.#runSpawn(toolCallId, params, signal, onUpdate, preAllocatedId); + } + + /** + * Resume an existing agent: revive it if parked, inject the follow-up + * assignment through the session's normal prompt path, and run it through + * the same yield/finalize pipeline as a spawn. The session stays alive + * (idle, TTL re-armed) afterwards. + */ + async #executeResume( + toolCallId: string, + params: TaskParams, + signal?: AbortSignal, + onUpdate?: AgentToolUpdateCallback, + ): Promise> { + const startTime = Date.now(); + const resumeId = params.resume!.trim(); + const simpleMode = this.#getTaskSimpleMode(); + const { customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); + const assignment = (params.assignment ?? "").trim(); + + let session: AgentSession; + try { + session = await AgentLifecycleManager.global().ensureLive(resumeId); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + throw new ToolError( + `Cannot resume "${resumeId}": ${message} Use \`irc\` op:"list" to see live agent ids; transcripts are readable at history://${resumeId}.`, + ); + } + + const agentName = AgentRegistry.global().get(resumeId)?.displayName ?? "task"; + const agentDef: AgentDefinition = getAgent(this.#discoveredAgents, agentName) ?? { + name: agentName, + description: "", + systemPrompt: "", + source: "bundled", + }; + + // Resumed output artifacts overwrite agent:// in the parent's + // artifacts dir; the transcript accretes in the session JSONL. + const sessionFile = this.session.getSessionFile(); + const artifactsDir = sessionFile ? sessionFile.slice(0, -6) : undefined; + + const result = await resumeSubprocess({ + session, + id: resumeId, + agent: agentDef, + task: renderSubagentUserPrompt(assignment, simpleMode), + assignment, + description: params.description, + index: 0, + parentToolCallId: toolCallId, + outputSchema: customSchemaEnabled ? params.schema : undefined, + signal, + onProgress: progress => { + onUpdate?.({ + content: [{ type: "text", text: `Resuming ${resumeId}...` }], + details: { + projectAgentsDir: null, + results: [], + totalDurationMs: Date.now() - startTime, + progress: [{ ...progress, recentTools: progress.recentTools.slice() }], + }, + }); + }, + eventBus: this.session.eventBus, + settings: this.session.settings, + artifactsDir, + }); + + return this.#buildResultPayload(result, null, Date.now() - startTime, ""); + } + + /** Spawn a fresh subagent and run it to completion. */ + async #runSpawn( + toolCallId: string, + params: TaskParams, + signal?: AbortSignal, + onUpdate?: AgentToolUpdateCallback, + preAllocatedId?: string, ): Promise> { const startTime = Date.now(); const { agents, projectAgentsDir } = await discoverAgents(this.session.cwd); - const { agent: agentName, context, schema: outputSchema } = params; + const agentName = params.agent ?? ""; const simpleMode = this.#getTaskSimpleMode(); - const { contextEnabled, customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); - const sharedContext = contextEnabled ? context?.trim() : undefined; + const { customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); + const outputSchema = params.schema; + const assignment = (params.assignment ?? "").trim(); const isolationMode = this.session.settings.get("task.isolation.mode"); const isolationRequested = "isolated" in params ? params.isolated === true : false; const isIsolated = isolationMode !== "none" && isolationRequested; const mergeMode = this.session.settings.get("task.isolation.merge"); const commitStyle = this.session.settings.get("task.isolation.commits"); - const maxConcurrency = this.session.settings.get("task.maxConcurrency"); const taskDepth = this.session.taskDepth ?? 0; const subagentLspEnabled = (this.session.enableLsp ?? true) && this.session.settings.get("task.enableLsp"); if (isolationMode === "none" && "isolated" in params) { return { - content: [ - { - type: "text", - text: "Task isolation is disabled.", - }, - ], - details: { - projectAgentsDir, - results: [], - totalDurationMs: 0, - }, + content: [{ type: "text", text: "Task isolation is disabled." }], + details: { projectAgentsDir, results: [], totalDurationMs: 0 }, }; } @@ -748,17 +692,8 @@ export class TaskTool implements AgentTool a.name).join(", ") || "none"; return { - content: [ - { - type: "text", - text: `Unknown agent "${agentName}". Available: ${available}`, - }, - ], - details: { - projectAgentsDir, - results: [], - totalDurationMs: 0, - }, + content: [{ type: "text", text: `Unknown agent "${agentName}". Available: ${available}` }], + details: { projectAgentsDir, results: [], totalDurationMs: 0 }, }; } @@ -773,11 +708,7 @@ export class TaskTool implements AgentTool 0 ? ` Available: ${enabled.join(", ")}` : ""}`, }, ], - details: { - projectAgentsDir, - results: [], - totalDurationMs: 0, - }, + details: { projectAgentsDir, results: [], totalDurationMs: 0 }, }; } @@ -817,38 +748,6 @@ export class TaskTool implements AgentTool(); - - // Update callback - const emitProgress = () => { - const progress = Array.from(progressMap.values()).sort((a, b) => a.index - b.index); - onUpdate?.({ - content: [{ type: "text", text: `Running ${params.tasks.length} agents...` }], - details: { - projectAgentsDir, - results: [], - totalDurationMs: Date.now() - startTime, - progress, - }, - }); - }; - try { // Check self-recursion prevention if (this.#blockedAgent && agentName === this.#blockedAgent) { @@ -928,11 +801,7 @@ export class TaskTool implements AgentTool null)); - uniqueIds = await outputManager.allocateBatch(tasks.map(t => t.id)); + agentId = await outputManager.allocate(params.id?.trim() || generateTaskName()); } - const tasksWithUniqueIds = tasks.map((t, i) => ({ ...t, id: uniqueIds[i] })); const availableSkills = [...(this.session.skills ?? [])]; // Resolve autoload skills from agent definition against available skills @@ -995,85 +849,101 @@ export class TaskTool implements AgentTool { + onUpdate?.({ + content: [{ type: "text", text: `Running agent ${agentId}...` }], + details: { + projectAgentsDir, + results: [], + totalDurationMs: Date.now() - startTime, + progress: [latestProgress], + }, }); - } + }; emitProgress(); - const runTask = async ( - task: (typeof tasksWithUniqueIds)[number], - index: number, - workerSignal?: AbortSignal, - ) => { + const buildCommitMessageFn = () => + commitStyle === "ai" && this.session.modelRegistry + ? async (diff: string) => { + return generateCommitMessage( + diff, + this.session.modelRegistry!, + this.session.settings, + this.session.getSessionId?.() ?? undefined, + ); + } + : undefined; + + const sharedRunOptions = { + cwd: this.session.cwd, + agent: effectiveAgent, + task: renderSubagentUserPrompt(assignment, simpleMode), + assignment, + planReference, + description: params.description, + index: 0, + parentToolCallId: toolCallId, + id: agentId, + taskDepth, + modelOverride, + parentActiveModelPattern, + thinkingLevel: thinkingLevelOverride, + outputSchema: effectiveOutputSchema, + sessionFile, + persistArtifacts: !!artifactsDir, + artifactsDir: effectiveArtifactsDir, + enableLsp: subagentLspEnabled, + signal, + eventBus: this.session.eventBus, + onProgress: (progress: AgentProgress) => { + // Shallow snapshot; recentTools is mutated in place by the + // executor, the rest is reassigned or immutable. A deep clone + // here cost O(extractedToolData) per progress event. + latestProgress = { ...progress, recentTools: progress.recentTools.slice() }; + emitProgress(); + }, + authStorage: this.session.authStorage, + modelRegistry: this.session.modelRegistry, + settings: this.session.settings, + mcpManager, + contextFiles, + skills: availableSkills, + autoloadSkills: resolvedAutoloadSkills, + workspaceTree: this.session.workspaceTree, + promptTemplates, + rules: this.session.rules, + preloadedExtensionPaths: this.session.extensionPaths, + preloadedCustomToolPaths: this.session.customToolPaths, + localProtocolOptions, + parentArtifactManager, + parentHindsightSessionState: this.session.getHindsightSessionState?.(), + parentMnemopiSessionState: this.session.getMnemopiSessionState?.(), + parentTelemetry: this.session.getTelemetry?.(), + parentEvalSessionId, + }; + + const runTask = async (): Promise => { if (!isIsolated) { - return runSubprocess({ - cwd: this.session.cwd, - agent: effectiveAgent, - task: renderSubagentUserPrompt(task.assignment, simpleMode), - assignment: task.assignment.trim(), - context: sharedContext, - planReference, - description: task.description, - index, - parentToolCallId: toolCallId, - id: task.id, - taskDepth, - modelOverride, - parentActiveModelPattern, - thinkingLevel: thinkingLevelOverride, - outputSchema: effectiveOutputSchema, - sessionFile, - persistArtifacts: !!artifactsDir, - artifactsDir: effectiveArtifactsDir, - contextFile: contextFilePath, - enableLsp: subagentLspEnabled, - signal: workerSignal ?? signal, - eventBus: this.session.eventBus, - onProgress: progress => { - // Shallow snapshot; recentTools is mutated in place by the - // executor, the rest is reassigned or immutable. A deep clone - // here cost O(extractedToolData) per progress event. - progressMap.set(index, { ...progress, recentTools: progress.recentTools.slice() }); - emitProgress(); - }, - authStorage: this.session.authStorage, - modelRegistry: this.session.modelRegistry, - settings: this.session.settings, - mcpManager, - contextFiles, - skills: availableSkills, - autoloadSkills: resolvedAutoloadSkills, - workspaceTree: this.session.workspaceTree, - promptTemplates, - rules: this.session.rules, - preloadedExtensionPaths: this.session.extensionPaths, - preloadedCustomToolPaths: this.session.customToolPaths, - localProtocolOptions, - parentArtifactManager, - parentHindsightSessionState: this.session.getHindsightSessionState?.(), - parentMnemopiSessionState: this.session.getMnemopiSessionState?.(), - parentTelemetry: this.session.getTelemetry?.(), - parentEvalSessionId, - }); + return runSubprocess(sharedRunOptions); } const taskStart = Date.now(); @@ -1084,73 +954,25 @@ export class TaskTool implements AgentTool { - progressMap.set(index, { ...progress, recentTools: progress.recentTools.slice() }); - emitProgress(); - }, - authStorage: this.session.authStorage, - modelRegistry: this.session.modelRegistry, - settings: this.session.settings, - mcpManager, - contextFiles, - skills: availableSkills, - autoloadSkills: resolvedAutoloadSkills, - workspaceTree: this.session.workspaceTree, - promptTemplates, - rules: this.session.rules, - localProtocolOptions, - parentArtifactManager, - parentHindsightSessionState: this.session.getHindsightSessionState?.(), - parentMnemopiSessionState: this.session.getMnemopiSessionState?.(), - parentTelemetry: this.session.getTelemetry?.(), - parentEvalSessionId, + preloadedExtensionPaths: undefined, + preloadedCustomToolPaths: undefined, }); if (mergeMode === "branch" && result.exitCode === 0) { try { - const commitMsg = - commitStyle === "ai" && this.session.modelRegistry - ? async (diff: string) => { - return generateCommitMessage( - diff, - this.session.modelRegistry!, - this.session.settings, - this.session.getSessionId?.() ?? undefined, - ); - } - : undefined; const commitResult = await commitToBranch( isolationDir, taskBaseline, - task.id, - task.description, - commitMsg, + agentId, + params.description, + buildCommitMessageFn(), ); return { ...result, @@ -1159,7 +981,7 @@ export class TaskTool implements AgentTool { - if (result !== undefined) { - return result; - } - const task = tasksWithUniqueIds[index]; - const assignment = task.assignment.trim(); - return { - index, - id: task.id, - agent: agentName, - agentSource: agent.source, - task: renderSubagentUserPrompt(assignment, simpleMode), - assignment, - description: task.description, - exitCode: 1, - output: "", - stderr: "Skipped (cancelled before start)", - truncated: false, - durationMs: 0, - tokens: 0, - modelOverride, - error: "Cancelled before start", - aborted: true, - abortReason: "Cancelled before start", - }; - }); - - // Aggregate usage from executor results (already accumulated incrementally) - const aggregatedUsage = createUsageTotals(); - let hasAggregatedUsage = false; - for (const result of results) { - if (result.usage) { - addUsageTotals(aggregatedUsage, result.usage); - hasAggregatedUsage = true; - } - } - - // Collect output paths (artifacts already written by executor in real-time) - const outputPaths: string[] = []; - const patchPaths: string[] = []; - for (const result of results) { - if (result.outputPath) { - outputPaths.push(result.outputPath); - } - if (result.patchPath) { - patchPaths.push(result.patchPath); - } - } + const result = await runTask(); let mergeSummary = ""; let changesApplied: boolean | null = null; let hadAnyChanges = false; - let mergedBranchesForNestedPatches: Set | null = null; + let mergedBranchForNestedPatches = false; if (isIsolated && repoRoot) { try { if (mergeMode === "branch") { - // Branch mode: merge task branches sequentially - const branchEntries = results - .filter(r => r.branchName && r.exitCode === 0 && !r.aborted) - .map(r => ({ branchName: r.branchName!, taskId: r.id, description: r.description })); - - if (branchEntries.length === 0) { + if (!result.branchName || result.exitCode !== 0 || result.aborted) { changesApplied = true; - hadAnyChanges = false; mergeSummary = "\n\nNo changes to apply."; } else { - const mergeResult = await mergeTaskBranches(repoRoot, branchEntries); - mergedBranchesForNestedPatches = new Set(mergeResult.merged); + const mergeResult = await mergeTaskBranches(repoRoot, [ + { branchName: result.branchName, taskId: result.id, description: result.description }, + ]); + mergedBranchForNestedPatches = mergeResult.merged.includes(result.branchName); changesApplied = mergeResult.failed.length === 0; hadAnyChanges = changesApplied && mergeResult.merged.length > 0; if (changesApplied) { mergeSummary = hadAnyChanges - ? `\n\nMerged ${mergeResult.merged.length} branch${mergeResult.merged.length === 1 ? "" : "es"}: ${mergeResult.merged.join(", ")}` + ? `\n\nMerged branch: ${result.branchName}` : "\n\nNo changes to apply."; } else { - const mergedPart = - mergeResult.merged.length > 0 ? `Merged: ${mergeResult.merged.join(", ")}.\n` : ""; - const failedPart = `Failed: ${mergeResult.failed.join(", ")}.`; const conflictPart = mergeResult.conflict ? `\nConflict: ${mergeResult.conflict}` : ""; - mergeSummary = `\n\nBranch merge failed. ${mergedPart}${failedPart}${conflictPart}\nUnmerged branches remain for manual resolution.`; + mergeSummary = `\n\nBranch merge failed: ${result.branchName}.${conflictPart}\nThe unmerged branch remains for manual resolution.`; } if (mergeResult.stashConflict) { mergeSummary += `\n\n${mergeResult.stashConflict}`; } - } - // Clean up merged branches (keep failed ones for manual resolution) - const allBranches = branchEntries.map(b => b.branchName); - if (changesApplied) { - await cleanupTaskBranches(repoRoot, allBranches); + // Clean up the merged branch (keep failed ones for manual resolution) + if (changesApplied) { + await cleanupTaskBranches(repoRoot, [result.branchName]); + } } } else { - // Patch mode: apply patches from successful tasks. Failed or - // aborted siblings must not block completed work from landing. - const successfulResults = results.filter(r => r.exitCode === 0 && !r.error && !r.aborted); - const patchesInOrder = successfulResults.map(result => result.patchPath).filter(Boolean) as string[]; - const missingPatch = successfulResults.some(result => !result.patchPath); - if (missingPatch) { + // Patch mode: apply the patch from a successful run. A failed or + // aborted run has nothing to apply and must not block the result. + const succeeded = result.exitCode === 0 && !result.error && !result.aborted; + if (!succeeded) { + changesApplied = true; + hadAnyChanges = false; + } else if (!result.patchPath) { changesApplied = false; hadAnyChanges = false; } else { - const patchStats = await Promise.all( - patchesInOrder.map(async patchPath => ({ - patchPath, - size: (await fs.stat(patchPath)).size, - })), - ); - const nonEmptyPatches = patchStats.filter(patch => patch.size > 0).map(patch => patch.patchPath); - if (nonEmptyPatches.length === 0) { + const patchText = await Bun.file(result.patchPath).text(); + if (!patchText.trim()) { changesApplied = true; hadAnyChanges = false; } else { - const patchTexts = await Promise.all( - nonEmptyPatches.map(async patchPath => Bun.file(patchPath).text()), - ); - const combinedPatch = patchTexts - .map(text => (text.endsWith("\n") ? text : `${text}\n`)) - .join(""); - if (!combinedPatch.trim()) { - changesApplied = true; - hadAnyChanges = false; - } else { - changesApplied = await git.patch.canApplyText(repoRoot, combinedPatch); - if (changesApplied) { - try { - await git.patch.applyText(repoRoot, combinedPatch); - hadAnyChanges = true; - } catch { - changesApplied = false; - hadAnyChanges = false; - } + const normalized = patchText.endsWith("\n") ? patchText : `${patchText}\n`; + changesApplied = await git.patch.canApplyText(repoRoot, normalized); + if (changesApplied) { + try { + await git.patch.applyText(repoRoot, normalized); + hadAnyChanges = true; + } catch { + changesApplied = false; + hadAnyChanges = false; } } } @@ -1359,10 +1102,7 @@ export class TaskTool implements AgentTool 0 - ? `\n\nPatch artifacts:\n${patchPaths.map(patch => `- ${patch}`).join("\n")}` - : ""; + const patchList = result.patchPath ? `\n\nPatch artifact:\n- ${result.patchPath}` : ""; mergeSummary = `\n\n${notification}${patchList}`; } } @@ -1376,34 +1116,15 @@ export class TaskTool implements AgentTool { - if (!r.nestedPatches || r.nestedPatches.length === 0 || r.exitCode !== 0 || r.aborted) { - return false; - } - if (mergeMode !== "branch") { - return true; - } - if (!r.branchName || !mergedBranchesForNestedPatches) { - return false; - } - return mergedBranchesForNestedPatches.has(r.branchName); - }) - .flatMap(r => r.nestedPatches!); - if (allNestedPatches.length > 0) { + const nestedPatches = result.nestedPatches ?? []; + const eligible = + nestedPatches.length > 0 && + result.exitCode === 0 && + !result.aborted && + (mergeMode !== "branch" || mergedBranchForNestedPatches); + if (eligible) { try { - const commitMsg = - commitStyle === "ai" && this.session.modelRegistry - ? async (diff: string) => { - return generateCommitMessage( - diff, - this.session.modelRegistry!, - this.session.settings, - this.session.getSessionId?.() ?? undefined, - ); - } - : undefined; - await applyNestedPatches(repoRoot, allNestedPatches, commitMsg); + await applyNestedPatches(repoRoot, nestedPatches, buildCommitMessageFn()); } catch { // Nested patch failures are non-fatal to the parent merge mergeSummary += @@ -1412,58 +1133,6 @@ export class TaskTool implements AgentTool r.aborted).length; - const successCount = results.filter(r => r.exitCode === 0 && !r.error && !r.aborted).length; - const totalDuration = Date.now() - startTime; - - const summaries = results.map(r => { - const status = r.aborted - ? "cancelled" - : r.exitCode === 0 && r.error - ? "merge failed" - : r.exitCode === 0 - ? "completed" - : `failed (exit ${r.exitCode})`; - const output = r.output.trim() || r.stderr.trim() || "(no output)"; - const outputCharCount = r.outputMeta?.charCount ?? output.length; - const fullOutputThreshold = 5000; - let preview = output; - let truncated = false; - if (outputCharCount > fullOutputThreshold) { - const slice = output.slice(0, fullOutputThreshold); - const lastNewline = slice.lastIndexOf("\n"); - preview = lastNewline >= 0 ? slice.slice(0, lastNewline) : slice; - truncated = true; - } - return { - agent: r.agent, - status, - id: r.id, - preview, - truncated, - meta: r.outputMeta - ? { - lineCount: r.outputMeta.lineCount, - charSize: formatBytes(r.outputMeta.charCount), - } - : undefined, - }; - }); - - const outputIds = results.filter(r => !r.aborted || r.output.trim()).map(r => `agent://${r.id}`); - const summary = prompt.render(taskSummaryTemplate, { - successCount, - totalCount: results.length, - cancelledCount, - hasCancelledNote: aborted && cancelledCount > 0, - duration: formatDuration(totalDuration), - summaries, - outputIds, - agentName, - mergeSummary, - }); - // Cleanup temp directory if used const shouldCleanupTempArtifacts = tempArtifactsDir && (!isIsolated || changesApplied === true || changesApplied === null); @@ -1471,25 +1140,65 @@ export class TaskTool implements AgentTool { + const status = result.aborted + ? "cancelled" + : result.exitCode === 0 && result.error + ? "merge failed" + : result.exitCode === 0 + ? "completed" + : `failed (exit ${result.exitCode})`; + const output = formatResultOutputFallback(result); + const outputCharCount = result.outputMeta?.charCount ?? output.length; + const fullOutputThreshold = 5000; + let preview = output; + let truncated = false; + if (outputCharCount > fullOutputThreshold) { + const slice = output.slice(0, fullOutputThreshold); + const lastNewline = slice.lastIndexOf("\n"); + preview = lastNewline >= 0 ? slice.slice(0, lastNewline) : slice; + truncated = true; + } + const summary = prompt.render(taskSummaryTemplate, { + agentName: result.agent, + id: result.id, + status, + duration: formatDuration(totalDurationMs), + preview, + truncated, + meta: result.outputMeta + ? { + lineCount: result.outputMeta.lineCount, + charSize: formatBytes(result.outputMeta.charCount), + } + : undefined, + mergeSummary, + }); + + return { + content: [{ type: "text", text: summary }], + details: { + projectAgentsDir, + results: [result], + totalDurationMs, + usage: result.usage, + outputPaths: result.outputPath ? [result.outputPath] : undefined, + }, + }; + } } diff --git a/packages/coding-agent/src/task/output-manager.ts b/packages/coding-agent/src/task/output-manager.ts index 46d6593c0..74fba7832 100644 --- a/packages/coding-agent/src/task/output-manager.ts +++ b/packages/coding-agent/src/task/output-manager.ts @@ -85,15 +85,4 @@ export class AgentOutputManager { await this.#ensureInitialized(); return this.#allocateUnique(id); } - - /** - * Allocate unique IDs for a batch of tasks. - * - * @param ids Array of requested IDs - * @returns Array of unique IDs in same order - */ - async allocateBatch(ids: string[]): Promise { - await this.#ensureInitialized(); - return ids.map(id => this.#allocateUnique(id)); - } } diff --git a/packages/coding-agent/src/task/render.ts b/packages/coding-agent/src/task/render.ts index fb881a2e6..556f59438 100644 --- a/packages/coding-agent/src/task/render.ts +++ b/packages/coding-agent/src/task/render.ts @@ -33,7 +33,7 @@ import { import { framedBlock, renderStatusLine } from "../tui"; import { repairDoubleEncodedJsonString } from "./repair-args"; import { subprocessToolRegistry } from "./subprocess-tool-registry"; -import type { AgentProgress, SingleResult, TaskItem, TaskParams, TaskToolDetails } from "./types"; +import type { AgentProgress, SingleResult, TaskParams, TaskToolDetails } from "./types"; /** * Get status icon for agent state. @@ -62,6 +62,7 @@ function appendAgentStats( line: string, opts: { toolCount?: number; + requests?: number; tokens: number; contextTokens?: number; contextWindow?: number; @@ -74,6 +75,9 @@ function appendAgentStats( if (opts.toolCount) { line += `${theme.sep.dot}${theme.fg("dim", `${formatNumber(opts.toolCount)} ${theme.icon.extensionTool}`)}`; } + if (opts.requests) { + line += `${theme.sep.dot}${theme.fg("dim", `${formatNumber(opts.requests)} req`)}`; + } // Current per-turn context — match the status line's `%/` gauge (e.g. `5.1%/1M`). if (opts.contextTokens && opts.contextTokens > 0) { const ctx = @@ -505,65 +509,57 @@ function formatOutputInline(data: unknown, theme: Theme, maxWidth = 80): string } /** - * Render the per-task list (`id` + ui `description`) for the streaming call - * preview. The args stream in token by token, so the array grows over time and - * trailing entries may be partially parsed — every field access is defensive. + * Render the call preview lines for the single spawned/resumed agent. The + * args stream in token by token, so every field access is defensive. */ -function renderTaskItemLines(tasks: TaskItem[] | undefined, expanded: boolean, theme: Theme): string[] { - const items = tasks ?? []; - if (items.length === 0) return []; - +function renderTaskCallLines(args: Partial | undefined, theme: Theme): string[] { + if (!args) return []; const bullet = theme.fg("dim", "•"); - const cap = expanded ? items.length : Math.min(items.length, 12); - const truncated = cap < items.length; - const lines: string[] = []; - for (let i = 0; i < cap; i++) { - const task = items[i] as Partial | undefined; - const rawId = task?.id?.trim(); - const idLabel = rawId ? formatTaskId(rawId) : `#${i + 1}`; - let line = `${bullet} ${theme.fg("accent", theme.bold(idLabel))}`; - const desc = task?.description?.trim(); + + const resume = typeof args.resume === "string" ? args.resume.trim() : ""; + const rawId = typeof args.id === "string" ? args.id.trim() : ""; + const idLabel = resume ? formatTaskId(resume) : rawId ? formatTaskId(rawId) : ""; + const desc = typeof args.description === "string" ? args.description.trim() : ""; + if (idLabel || desc) { + let line = `${bullet} ${theme.fg("accent", theme.bold(idLabel || "agent"))}`; if (desc) { line += `: ${theme.fg("muted", truncateToWidth(replaceTabs(desc), 64))}`; } lines.push(line); } - if (truncated) { - lines.push(`${bullet} ${theme.fg("dim", formatMoreItems(items.length - cap, "agent"))}`); - } return lines; } -/** - * Build the shared-context section (the `# Goal / # Constraints` background - * passed to every subagent). Rendered in both the streaming call preview and - * the merged result frame so the brief stays visible for the whole task - * lifecycle — not just until the first progress snapshot replaces the call view. - */ -type TaskRenderSection = { lines: readonly string[] }; -type ContextSectionRenderer = (width: number) => TaskRenderSection; +/** One renderable frame section: optional label, body rows, leading divider. */ +type TaskRenderSection = { label?: string; lines: readonly string[]; separator?: boolean }; +type AssignmentSectionRenderer = (width: number) => TaskRenderSection; // Default output-block layout is: left border + one-cell content inset + right // border. Render markdown at that inner width so the output block does not need -// to rewrap already-rendered context lines. -const CONTEXT_FRAME_INSET = 3; +// to rewrap already-rendered assignment lines. +const ASSIGNMENT_FRAME_INSET = 3; -function contextMarkdownWidth(frameWidth: number): number { - return Math.max(1, frameWidth - CONTEXT_FRAME_INSET); -} - -function createContextSectionRenderer(args: TaskParams | undefined, theme: Theme): ContextSectionRenderer | undefined { +/** + * Build the assignment section (the markdown brief handed to the subagent). + * Rendered in both the streaming call preview and the result frame so the + * brief stays visible for the whole task lifecycle — not just until the first + * progress snapshot replaces the call view. + */ +function createAssignmentSectionRenderer( + args: Partial | undefined, + theme: Theme, +): AssignmentSectionRenderer | undefined { // `renderResult` receives the raw tool args (unlike `renderCall`, which is - // fed through `repairTaskParams`), so undo any per-field double-encoding here - // too. The repair is idempotent on already-clean text. - const context = repairDoubleEncodedJsonString(args?.context ?? "").trim(); - if (!context) return undefined; + // fed through `repairTaskParams`), so undo any per-field double-encoding + // here too. The repair is idempotent on already-clean text. + const assignment = repairDoubleEncodedJsonString(typeof args?.assignment === "string" ? args.assignment : "").trim(); + if (!assignment) return undefined; - const markdown = new Markdown(context, 0, 0, getMarkdownTheme(), { + const markdown = new Markdown(assignment, 0, 0, getMarkdownTheme(), { color: text => theme.fg("muted", text), }); - return width => ({ lines: markdown.render(contextMarkdownWidth(width)) }); + return width => ({ lines: markdown.render(Math.max(1, width - ASSIGNMENT_FRAME_INSET)) }); } /** @@ -575,22 +571,23 @@ export function renderCall( theme: Theme, ): Component { const showIsolated = "isolated" in args && args.isolated === true; - const header = renderStatusLine({ icon: "pending", title: "Task", description: args.agent }, theme); - const contextSectionRenderer = createContextSectionRenderer(args, theme); + const resume = typeof args.resume === "string" && args.resume.trim() ? args.resume.trim() : undefined; + const headerDescription = resume ? `resume ${formatTaskId(resume)}` : args.agent; + const header = renderStatusLine({ icon: "pending", title: "Task", description: headerDescription }, theme); + const assignmentSection = createAssignmentSectionRenderer(args, theme); return framedBlock(theme, width => { const sections: Array<{ label?: string; lines: readonly string[]; separator?: boolean }> = []; - if (contextSectionRenderer) sections.push(contextSectionRenderer(width)); - - // The per-task preview list only exists to surface dispatched agents while - // the call args stream in. Once a result snapshot exists, `renderResult` - // draws the same agents as progress/result lines, so showing the Tasks - // section here would just repeat the count the result frame already shows. + // The call preview only exists to surface the dispatched agent while the + // args stream in. Once a result snapshot exists, `renderResult` draws the + // same agent (and the assignment brief) itself, so showing it here would + // repeat what the result frame already shows. if (!options.renderContext?.hasResult) { sections.push({ separator: true, - lines: renderTaskItemLines(args.tasks, options.expanded, theme), + lines: renderTaskCallLines(args, theme), }); + if (assignmentSection) sections.push(assignmentSection(width)); } return { @@ -631,8 +628,12 @@ function renderAgentProgress( const titlePart = description ? `${theme.bold(displayId)}: ${description}` : displayId; const indent = prefix ? `${prefix} ` : ""; let statusLine: string; - if (progress.status === "running") { - const bullet = theme.styledSymbol("status.done", "text"); + if (progress.status === "running" || progress.status === "pending") { + // Live (or queued) agents shimmer their description so the row reads as + // in-flight even after the block freezes — the async spawn result keeps + // the agent on "pending" while the detached job runs. + const bullet = + progress.status === "running" ? theme.styledSymbol("status.done", "text") : theme.fg(iconColor, icon); const name = theme.fg("accent", description ? theme.bold(displayId) : displayId); statusLine = `${indent}${bullet} ${name}`; if (description) { @@ -945,6 +946,7 @@ function renderAgentResult( statusLine, { tokens: result.tokens, + requests: result.requests, contextTokens: result.contextTokens, contextWindow: result.contextWindow, cost: result.usage?.cost.total ?? 0, @@ -1073,9 +1075,11 @@ function renderAgentResult( } /** - * Order live progress entries so finished agents render first and unfinished - * (pending/running) ones stay pinned at the bottom as tasks complete. Stable - * within each group, so agents keep their dispatch order. + * Order live progress entries so finished agents render first — sorted by + * runtime ascending, matching {@link orderResultsForDisplay} — while + * unfinished (pending/running) ones stay pinned at the bottom in dispatch + * order. Because a finished agent's runtime is fixed, finalization renders + * the same order and rows never reshuffle. */ function orderProgressForDisplay(progress: readonly AgentProgress[]): AgentProgress[] { const finished: AgentProgress[] = []; @@ -1083,9 +1087,19 @@ function orderProgressForDisplay(progress: readonly AgentProgress[]): AgentProgr for (const p of progress) { (p.status === "pending" || p.status === "running" ? unfinished : finished).push(p); } + finished.sort((a, b) => a.durationMs - b.durationMs || a.index - b.index); return finished.concat(unfinished); } +/** + * Order finalized results by runtime ascending (tie-break: dispatch index) so + * the finalized list matches the live-progress order produced by + * {@link orderProgressForDisplay}. + */ +function orderResultsForDisplay(results: readonly SingleResult[]): SingleResult[] { + return [...results].sort((a, b) => a.durationMs - b.durationMs || a.index - b.index); +} + /** * Render the tool result. */ @@ -1097,25 +1111,27 @@ export function renderResult( ): Component { const fallbackText = result.content.find(c => c.type === "text")?.text ?? ""; const details = result.details; - const contextSectionRenderer = createContextSectionRenderer(args, theme); + const resumeLabel = + typeof args?.resume === "string" && args.resume.trim() ? `resume ${formatTaskId(args.resume.trim())}` : undefined; + const assignmentSection = createAssignmentSectionRenderer(args, theme); if (!details) { const text = result.content.find(c => c.type === "text")?.text || ""; const errored = result.isError === true; const header = errored - ? renderStatusLine({ icon: "error", title: "Task", description: args?.agent }, theme) + ? renderStatusLine({ icon: "error", title: "Task", description: resumeLabel ?? args?.agent }, theme) : renderStatusLine( { iconOverride: theme.styledSymbol("status.done", "accent"), title: "Task", - description: args?.agent, + description: resumeLabel ?? args?.agent, }, theme, ); return framedBlock(theme, width => ({ header, sections: [ - ...(contextSectionRenderer ? [contextSectionRenderer(width)] : []), + ...(assignmentSection ? [assignmentSection(width)] : []), ...(text ? [{ separator: true, lines: [theme.fg("dim", truncateToWidth(text, width))] }] : []), ], state: errored ? "error" : "success", @@ -1131,10 +1147,9 @@ export function renderResult( const isError = aborted || failed; const agentCount = hasResults ? details.results.length : (details.progress?.length ?? 0); const icon: ToolUIStatus = options.isPartial ? "running" : isError ? "error" : mergeFailed ? "warning" : "success"; - // Surface the dispatched agent type (e.g. `Reviewer`) alongside the count so - // the header reads `Task 16 agents: Reviewer`. All tasks in one call share a - // single `agent` type (top-level param), so one label covers the whole batch. - const agentName = args?.agent?.trim(); + // Surface the dispatched agent type (e.g. `Reviewer`) or the resumed agent + // id alongside the count so the header reads `Task 1 agent: Reviewer`. + const agentName = resumeLabel ?? args?.agent?.trim(); const countLabel = agentCount > 0 ? `${agentCount} ${agentCount === 1 ? "agent" : "agents"}` : undefined; const metaLabel = countLabel ? (agentName ? `${countLabel}: ${agentName}` : countLabel) : agentName; const header = renderStatusLine( @@ -1158,7 +1173,7 @@ export function renderResult( lines.push(...renderAgentProgress(progress, "", " ", expanded, theme, spinnerFrame)); }); } else if (details.results && details.results.length > 0) { - details.results.forEach(res => { + orderResultsForDisplay(details.results).forEach(res => { lines.push(...renderAgentResult(res, "", " ", expanded, theme)); }); @@ -1171,6 +1186,8 @@ export function renderResult( if (successCount > 0) summaryParts.push(theme.fg("success", `${successCount} succeeded`)); if (mergeFailedCount > 0) summaryParts.push(theme.fg("warning", `${mergeFailedCount} merge failed`)); if (failCount > 0) summaryParts.push(theme.fg("error", `${failCount} failed`)); + const totalRequests = details.results.reduce((sum, r) => sum + (r.requests ?? 0), 0); + if (totalRequests > 0) summaryParts.push(theme.fg("dim", `${formatNumber(totalRequests)} req`)); summaryParts.push(theme.fg("dim", formatDuration(details.totalDurationMs))); // Wrap the run summary in the theme's bracket glyphs (dim chrome, colored // counts) to match the bash tool's `[Wall: … | Exit: …]` footer. @@ -1189,7 +1206,7 @@ export function renderResult( return { header, sections: [ - ...(contextSectionRenderer ? [contextSectionRenderer(width)] : []), + ...(assignmentSection ? [assignmentSection(width)] : []), { separator: true, lines: [theme.fg("dim", truncateToWidth(text, width))] }, ], state, @@ -1219,7 +1236,7 @@ export function renderResult( return { header, sections: [ - ...(contextSectionRenderer ? [contextSectionRenderer(width)] : []), + ...(assignmentSection ? [assignmentSection(width)] : []), ...(lines.length > 0 ? [{ separator: true, lines }] : []), ], state, @@ -1252,8 +1269,9 @@ function renderNestedTaskResults(detailsList: TaskToolDetails[], expanded: boole const lines: string[] = []; for (const details of detailsList) { if (!details.results || details.results.length === 0) continue; - details.results.forEach((result, index) => { - const { prefix, continuePrefix } = nestedMarkers(index === details.results.length - 1, theme); + const ordered = orderResultsForDisplay(details.results); + ordered.forEach((result, index) => { + const { prefix, continuePrefix } = nestedMarkers(index === ordered.length - 1, theme); lines.push(...renderAgentResult(result, prefix, continuePrefix, expanded, theme)); }); } @@ -1275,8 +1293,9 @@ function renderNestedTaskTree( for (const details of detailsList) { const hasResults = Boolean(details.results && details.results.length > 0); if (hasResults) { - details.results.forEach((result, index) => { - const { prefix, continuePrefix } = nestedMarkers(index === details.results.length - 1, theme); + const ordered = orderResultsForDisplay(details.results); + ordered.forEach((result, index) => { + const { prefix, continuePrefix } = nestedMarkers(index === ordered.length - 1, theme); lines.push(...renderAgentResult(result, prefix, continuePrefix, expanded, theme)); }); continue; diff --git a/packages/coding-agent/src/task/repair-args.ts b/packages/coding-agent/src/task/repair-args.ts index 62cc8f7a7..ec52a0000 100644 --- a/packages/coding-agent/src/task/repair-args.ts +++ b/packages/coding-agent/src/task/repair-args.ts @@ -2,7 +2,7 @@ * Repair double-encoded JSON string arguments for the task tool. * * Models occasionally JSON-escape a string value twice when emitting a - * `task` tool call, so a `context`/`assignment` that should read + * `task` tool call, so an `assignment` that should read * * # Role * You are a judge … "describe this" … return — @@ -24,11 +24,11 @@ * string. * * This is deliberately scoped to the task tool's natural-language fields - * (`context`, `assignment`, `description`). It is NOT applied to code-bearing + * (`assignment`, `description`). It is NOT applied to code-bearing * tools (write/edit/bash/search), where a backslash or quote is load-bearing * and a false-positive unescape would silently corrupt a file or command. */ -import type { TaskItem, TaskParams } from "./types"; +import type { TaskParams } from "./types"; /** A backslash that escapes a structural char — `\"`, `\\`, `\/`, or `\uXXXX`. */ const STRUCTURAL_ESCAPE = /\\(?:["\\/]|u[0-9a-fA-F]{4})/; @@ -78,40 +78,21 @@ export function repairDoubleEncodedJsonString(value: string): string { return typeof decoded === "string" && decoded !== value ? decoded : value; } -/** Repair a single (possibly partial) task item's prose fields. */ -function repairTaskItem(task: TaskItem): TaskItem { - if (task === null || typeof task !== "object") return task; - const assignment = - typeof task.assignment === "string" ? repairDoubleEncodedJsonString(task.assignment) : task.assignment; - const description = - typeof task.description === "string" ? repairDoubleEncodedJsonString(task.description) : task.description; - if (assignment === task.assignment && description === task.description) return task; - return { ...task, assignment, description }; -} - /** - * Repair double-encoded prose in task-tool params (`context` and each task's - * `assignment`/`description`). Returns the same reference when nothing changed - * so callers can cheaply skip work. Defensive against partially-streamed args - * (missing/undefined fields, partial task arrays) so it is safe on the render - * path as well as on execution. + * Repair double-encoded prose in task-tool params (`assignment` and + * `description`). Returns the same reference when nothing changed so callers + * can cheaply skip work. Defensive against partially-streamed args + * (missing/undefined fields) so it is safe on the render path as well as on + * execution. */ export function repairTaskParams(params: TaskParams): TaskParams { if (params === null || typeof params !== "object") return params; - const context = typeof params.context === "string" ? repairDoubleEncodedJsonString(params.context) : params.context; + const assignment = + typeof params.assignment === "string" ? repairDoubleEncodedJsonString(params.assignment) : params.assignment; + const description = + typeof params.description === "string" ? repairDoubleEncodedJsonString(params.description) : params.description; - let tasks = params.tasks; - if (Array.isArray(params.tasks)) { - let changed = false; - const repaired = params.tasks.map(task => { - const next = repairTaskItem(task); - if (next !== task) changed = true; - return next; - }); - if (changed) tasks = repaired; - } - - if (context === params.context && tasks === params.tasks) return params; - return { ...params, context, tasks }; + if (assignment === params.assignment && description === params.description) return params; + return { ...params, assignment, description }; } diff --git a/packages/coding-agent/src/task/simple-mode.ts b/packages/coding-agent/src/task/simple-mode.ts index fc4269f13..22c3c607d 100644 --- a/packages/coding-agent/src/task/simple-mode.ts +++ b/packages/coding-agent/src/task/simple-mode.ts @@ -3,21 +3,17 @@ export const TASK_SIMPLE_MODES = ["default", "schema-free", "independent"] as co export type TaskSimpleMode = (typeof TASK_SIMPLE_MODES)[number]; interface TaskSimpleModeCapabilities { - contextEnabled: boolean; customSchemaEnabled: boolean; } const TASK_SIMPLE_MODE_CAPABILITIES: Record = { default: { - contextEnabled: true, customSchemaEnabled: true, }, "schema-free": { - contextEnabled: true, customSchemaEnabled: false, }, independent: { - contextEnabled: false, customSchemaEnabled: false, }, }; diff --git a/packages/coding-agent/src/task/types.ts b/packages/coding-agent/src/task/types.ts index 5b535e649..01b84a1e5 100644 --- a/packages/coding-agent/src/task/types.ts +++ b/packages/coding-agent/src/task/types.ts @@ -66,34 +66,16 @@ export interface SubagentLifecyclePayload { index: number; } -const assignmentDescription = "per-task instructions; self-contained"; - -const createTaskItemSchema = (_contextEnabled: boolean) => - z.object({ - id: z.string().max(48).describe("camelcase identifier"), - description: z.string().describe("ui label, not seen by subagent"), - assignment: z.string().describe(assignmentDescription), - }); - -/** Single task item for parallel execution (default shape with context enabled). */ -export const taskItemSchema = createTaskItemSchema(true); -export type TaskItem = z.infer; - -const createTaskSchema = (options: { isolationEnabled: boolean; simpleMode: TaskSimpleMode }) => { - const { contextEnabled, customSchemaEnabled } = getTaskSimpleModeCapabilities(options.simpleMode); - const itemSchema = createTaskItemSchema(contextEnabled); - +const createTaskSchema = (options: { isolationEnabled: boolean; customSchemaEnabled: boolean }) => { let schema = z.object({ - agent: z.string().describe("agent type"), - tasks: z.array(itemSchema).describe("tasks to execute in parallel"), + agent: z.string().optional().describe("agent type; omit when resume is set"), + id: z.string().max(48).optional().describe("stable agent id; default generated"), + description: z.string().optional().describe("ui label, not seen by subagent"), + assignment: z.string().describe("the work; self-contained instructions"), + resume: z.string().optional().describe("existing agent id: revive and continue instead of spawning"), }); - if (contextEnabled) { - schema = schema.extend({ - context: z.string().optional().describe("shared background prepended to each assignment"), - }); - } - if (customSchemaEnabled) { + if (options.customSchemaEnabled) { schema = schema.extend({ schema: z.string().optional().describe("jtd schema for expected response shape"), }); @@ -108,19 +90,15 @@ const createTaskSchema = (options: { isolationEnabled: boolean; simpleMode: Task return schema; }; -export const taskSchema = createTaskSchema({ isolationEnabled: true, simpleMode: "default" }); -export const taskSchemaNoIsolation = createTaskSchema({ isolationEnabled: false, simpleMode: "default" }); -const taskSchemaSchemaFree = createTaskSchema({ isolationEnabled: true, simpleMode: "schema-free" }); -const taskSchemaSchemaFreeNoIsolation = createTaskSchema({ isolationEnabled: false, simpleMode: "schema-free" }); -const taskSchemaIndependent = createTaskSchema({ isolationEnabled: true, simpleMode: "independent" }); -const taskSchemaIndependentNoIsolation = createTaskSchema({ isolationEnabled: false, simpleMode: "independent" }); +export const taskSchema = createTaskSchema({ isolationEnabled: true, customSchemaEnabled: true }); +const taskSchemaNoIsolation = createTaskSchema({ isolationEnabled: false, customSchemaEnabled: true }); +const taskSchemaSchemaFree = createTaskSchema({ isolationEnabled: true, customSchemaEnabled: false }); +const taskSchemaSchemaFreeNoIsolation = createTaskSchema({ isolationEnabled: false, customSchemaEnabled: false }); const ALL_TASK_SCHEMAS = [ taskSchema, taskSchemaNoIsolation, taskSchemaSchemaFree, taskSchemaSchemaFreeNoIsolation, - taskSchemaIndependent, - taskSchemaIndependentNoIsolation, ] as const; type DynamicTaskSchema = (typeof ALL_TASK_SCHEMAS)[number]; @@ -129,22 +107,28 @@ export type TaskSchema = typeof taskSchema; export type TaskToolSchemaInstance = DynamicTaskSchema; export function getTaskSchema(options: { isolationEnabled: boolean; simpleMode: TaskSimpleMode }): DynamicTaskSchema { - switch (options.simpleMode) { - case "schema-free": - return options.isolationEnabled ? taskSchemaSchemaFree : taskSchemaSchemaFreeNoIsolation; - case "independent": - return options.isolationEnabled ? taskSchemaIndependent : taskSchemaIndependentNoIsolation; - default: - return options.isolationEnabled ? taskSchema : taskSchemaNoIsolation; + const { customSchemaEnabled } = getTaskSimpleModeCapabilities(options.simpleMode); + if (customSchemaEnabled) { + return options.isolationEnabled ? taskSchema : taskSchemaNoIsolation; } + return options.isolationEnabled ? taskSchemaSchemaFree : taskSchemaSchemaFreeNoIsolation; } export interface TaskParams { - agent: string; - context?: string; + /** Agent type; required unless `resume` is set. */ + agent?: string; + /** Stable agent id; default = generated AdjectiveNoun. */ + id?: string; + /** UI label, not seen by the subagent. */ + description?: string; + /** The work; required. */ + assignment?: string; + /** JTD schema for the expected yield shape; unchanged semantics. */ schema?: string; - tasks: TaskItem[]; + /** Run in an isolated worktree; isolated agents are NOT resumable. */ isolated?: boolean; + /** Existing agent id: revive + follow-up instead of spawn. */ + resume?: string; } /** A code review finding reported by the reviewer agent */ @@ -206,6 +190,8 @@ export interface AgentProgress { recentTools: Array<{ tool: string; args: string; endMs: number }>; recentOutput: string[]; toolCount: number; + /** Count of assistant requests (assistant message_end events) across the run. Drives the soft request budget guard. */ + requests: number; /** Cumulative input + output + cacheWrite tokens across all turns. Excludes cacheRead (re-reads cached context every turn, making cumulative sum misleading). */ tokens: number; /** @@ -276,6 +262,8 @@ export interface SingleResult { durationMs: number; /** Cumulative input + output + cacheWrite tokens across all turns. Excludes cacheRead (re-reads cached context every turn, making cumulative sum misleading). */ tokens: number; + /** Count of assistant requests (assistant message_end events) across the run. */ + requests: number; /** Latest per-turn context size at task completion. See `AgentProgress.contextTokens`. */ contextTokens?: number; /** Model's context window in tokens, when known. */ diff --git a/packages/coding-agent/test/eval/agent-bridge.test.ts b/packages/coding-agent/test/eval/agent-bridge.test.ts index 832dd7fba..492087753 100644 --- a/packages/coding-agent/test/eval/agent-bridge.test.ts +++ b/packages/coding-agent/test/eval/agent-bridge.test.ts @@ -21,6 +21,7 @@ function createResult(): SingleResult { truncated: false, durationMs: 1, tokens: 0, + requests: 0, }; } diff --git a/packages/coding-agent/test/rpc-subagents.test.ts b/packages/coding-agent/test/rpc-subagents.test.ts index 65b89a31b..133373bdd 100644 --- a/packages/coding-agent/test/rpc-subagents.test.ts +++ b/packages/coding-agent/test/rpc-subagents.test.ts @@ -43,6 +43,7 @@ function createProgress(overrides: Partial = {}): AgentProgress { recentTools: [], recentOutput: [], toolCount: 0, + requests: 0, tokens: 0, cost: 0, durationMs: 0, diff --git a/packages/coding-agent/test/streaming-preview-height.test.ts b/packages/coding-agent/test/streaming-preview-height.test.ts index 7bfae7adf..acf37be34 100644 --- a/packages/coding-agent/test/streaming-preview-height.test.ts +++ b/packages/coding-agent/test/streaming-preview-height.test.ts @@ -372,19 +372,26 @@ describe("streaming tool call preview height (bounded across renderers)", () => } }, 30_000); - test("task pending preview preserves full multiline context", () => { + test("task pending preview stays bounded with a long multiline assignment", () => { + // CONTRACT CHANGE with the single-spawn task rework: the old uncapped + // multi-task `context` rendering is gone with the field. The pending + // preview now intentionally bounds the assignment (first line + a + // "more lines" marker when collapsed; 12 lines when expanded), like + // bash/ssh, so a long assignment can no longer strand the block top. const longLines = Array.from({ length: 80 }, (_, i) => `line-${i}`); const { lines, text } = renderPending("task", { agent: "task", - context: longLines.join("\n"), - tasks: [{ id: "alpha", description: "preview" }], + id: "alpha", + description: "preview", + assignment: longLines.join("\n"), }); - expect(lines.length, "task preview should not be capped").toBeGreaterThan(80); + expect(lines.length, "task preview should stay bounded").toBeLessThan(20); + expect(text).toContain("preview"); expect(text).toContain("line-0"); - expect(text).toContain("line-40"); - expect(text).toContain("line-79"); - expect(text).not.toMatch(/more lines/); + expect(text).not.toContain("line-40"); + expect(text).not.toContain("line-79"); + expect(text, "task preview should advertise truncation").toMatch(/more lines/); }); test("eval pending preview preserves full code (never collapsed)", () => { diff --git a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts index 1b308036a..7f69af243 100644 --- a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts +++ b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts @@ -197,7 +197,7 @@ describe("runSubprocess yield reminders", () => { expect(createAgentSessionSpy).toHaveBeenCalledTimes(1); }); - it("renders shared task context in subagent system prompt before now", async () => { + it("splices the subagent role prompt before the trailing system section", async () => { let userPrompt = ""; const session = createMockSession(({ text, emit }) => { userPrompt = text; @@ -217,7 +217,6 @@ describe("runSubprocess yield reminders", () => { await runSubprocess({ ...baseOptions, id: "subagent-context-system", - context: "Shared task background", task: "Your assignment is below.\nBe thorough and complete fully before yielding.\n\nDo the task.", }); @@ -229,11 +228,12 @@ describe("runSubprocess yield reminders", () => { expect(systemPrompt).toHaveLength(4); expect(systemPrompt?.[0]).toBe("system"); expect(systemPrompt?.[1]).toBe("project"); - expect(systemPrompt?.[2]).toMatch(/CONTEXT\n=+\n\nShared task background/); expect(systemPrompt?.[2]).toMatch(/ROLE\n=+\n\ntest/); + // The parent-conversation CONTEXT section is gone: subagents get their + // background inside the assignment (or a local:// file), never a dump. + expect(systemPrompt?.[2]).not.toMatch(/CONTEXT\n=+/); expect(systemPrompt?.[3]).toBe("now"); expect(userPrompt).not.toMatch(/CONTEXT\n=+/); - expect(userPrompt).not.toContain("Shared task background"); }); it("sends reminder prompt when subagent stops without yield", async () => { @@ -586,7 +586,7 @@ describe("runSubprocess yield reminders", () => { expect(result.aborted).toBe(true); expect(errorSpy).not.toHaveBeenCalledWith("Subagent prompt failed", expect.anything()); - expect(debugSpy).toHaveBeenCalledWith("Subagent prompt aborted", expect.anything()); + expect(debugSpy).toHaveBeenCalledWith("Subagent prompt aborted"); }); }); diff --git a/packages/coding-agent/test/task/output-manager.test.ts b/packages/coding-agent/test/task/output-manager.test.ts index 50fef2440..b374105fe 100644 --- a/packages/coding-agent/test/task/output-manager.test.ts +++ b/packages/coding-agent/test/task/output-manager.test.ts @@ -19,10 +19,14 @@ describe("AgentOutputManager", () => { expect(await mgr.allocate("Bob")).toBe("Bob"); }); - it("de-duplicates within a batch while preserving order", async () => { + it("de-duplicates repeated names while preserving order", async () => { const mgr = new AgentOutputManager(() => null); - expect(await mgr.allocateBatch(["Auth", "Auth", "Api", "Auth"])).toEqual(["Auth", "Auth-2", "Api", "Auth-3"]); + const ids: string[] = []; + for (const name of ["Auth", "Auth", "Api", "Auth"]) { + ids.push(await mgr.allocate(name)); + } + expect(ids).toEqual(["Auth", "Auth-2", "Api", "Auth-3"]); }); it("nests ids under a parent prefix and still suffixes repeats", async () => { diff --git a/packages/coding-agent/test/task/render-call.test.ts b/packages/coding-agent/test/task/render-call.test.ts index 4ea12d880..f666239de 100644 --- a/packages/coding-agent/test/task/render-call.test.ts +++ b/packages/coding-agent/test/task/render-call.test.ts @@ -1,7 +1,7 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { getThemeByName, setThemeInstance, type Theme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import type { TaskParams, TaskToolDetails } from "@oh-my-pi/pi-coding-agent/task"; +import type { TaskParams } from "@oh-my-pi/pi-coding-agent/task"; import { taskToolRenderer } from "@oh-my-pi/pi-coding-agent/task/render"; describe("task renderer: streaming call preview", () => { @@ -25,100 +25,62 @@ describe("task renderer: streaming call preview", () => { return Bun.stripANSI(component.render(160).join("\n")); } - function renderCompleted(args: TaskParams): string { - const details: TaskToolDetails = { - projectAgentsDir: null, - totalDurationMs: 12, - results: [ - { - index: 0, - id: "Only", - agent: args.agent, - agentSource: "bundled", - task: "Render the shared context", - exitCode: 0, - output: "Done.", - stderr: "", - truncated: false, - durationMs: 12, - tokens: 1, - }, - ], - }; - const component = taskToolRenderer.renderResult( - { content: [{ type: "text", text: "1 agent completed." }], details }, - { expanded: false, isPartial: false }, - theme, - args, - ); - return Bun.stripANSI(component.render(160).join("\n")); - } - - // The preview must surface each agent's id + ui description so the user can - // see which agents are being dispatched, not a bare "N agents" count. - it("lists each task's id and description instead of only a count", () => { + // The preview must surface the agent id + ui description so the user can + // see what is being dispatched while args stream in. + it("shows the agent id, description, and assignment preview", () => { const args: TaskParams = { agent: "reviewer", - tasks: [ - { id: "ReviewAuth", description: "Audit the auth module", assignment: "..." }, - { id: "ReviewDb", description: "Audit the db layer", assignment: "..." }, - ], + id: "ReviewAuth", + description: "Audit the auth module", + assignment: "Review packages/server/src/auth for missing 401 handling.\nReport findings.", }; const out = render(args); + expect(out).toContain("reviewer"); expect(out).toContain("ReviewAuth"); expect(out).toContain("Audit the auth module"); - expect(out).toContain("ReviewDb"); - expect(out).toContain("Audit the db layer"); - // The per-task list stands on its own — neither the redundant "Tasks (N)" - // section label nor the old flat "N agents" line is drawn. - expect(out).not.toContain("Tasks ("); - expect(out).not.toContain("2 agents"); + expect(out).toContain("Review packages/server/src/auth for missing 401 handling."); }); - it("renders a partially-streamed entry without a description and missing trailing entry", () => { + it("renders partially-streamed args without crashing", () => { const args = { agent: "task", - // Trailing entry mimics streaming JSON: id arrived, description not yet, - // plus a not-yet-materialized slot. - tasks: [{ id: "First", description: "Do the first thing", assignment: "..." }, { id: "Second" }, undefined], + id: "First", + // description/assignment not yet arrived. } as unknown as TaskParams; const out = render(args); expect(out).toContain("First"); - expect(out).toContain("Do the first thing"); - expect(out).toContain("Second"); - // Missing-id slot falls back to a positional placeholder rather than crashing. - expect(out).toContain("#3"); - expect(out).not.toContain("Tasks ("); + expect(out).toContain("task"); }); - it("caps the collapsed list and reports the overflow as agents", () => { - const tasks = Array.from({ length: 15 }, (_, i) => ({ - id: `Agent${i + 1}`, - description: `Task ${i + 1}`, - assignment: "...", - })); - const args: TaskParams = { agent: "task", tasks }; + it("always renders the full assignment markdown, collapsed or expanded", () => { + const assignmentLines = Array.from({ length: 6 }, (_, i) => `Step ${i + 1}: do the thing.`); + const args: TaskParams = { + agent: "task", + id: "Worker", + assignment: assignmentLines.join("\n"), + }; + // The assignment is the brief handed to the subagent; it renders as + // markdown in full regardless of the expanded toggle. const collapsed = render(args, false); - expect(collapsed).toContain("Agent1"); - expect(collapsed).toContain("Agent12"); - expect(collapsed).not.toContain("Agent13"); - expect(collapsed).toContain("3 more agents"); + expect(collapsed).toContain("Step 1"); + expect(collapsed).toContain("Step 6"); const expanded = render(args, true); - expect(expanded).toContain("Agent13"); - expect(expanded).toContain("Agent15"); - expect(expanded).not.toContain("more agents"); + expect(expanded).toContain("Step 1"); + expect(expanded).toContain("Step 6"); }); it("surfaces the isolation flag in the header bar", () => { const args: TaskParams = { agent: "task", isolated: true, - tasks: [{ id: "Only", description: "Single task", assignment: "..." }], + id: "Only", + description: "Single task", + assignment: "...", }; const out = render(args); const lines = out.split("\n"); @@ -129,34 +91,28 @@ describe("task renderer: streaming call preview", () => { expect(lines[0]).toContain("isolated"); }); - it("renders shared context as markdown in call and result frames", () => { + it("labels resume calls with the resumed agent id", () => { const args: TaskParams = { - agent: "task", - context: ["# Goal", "Fix **rendering**.", "", "# Constraints", "- Keep `task` visible"].join("\n"), - tasks: [{ id: "Only", description: "Single task", assignment: "..." }], + resume: "AuthLoader", + assignment: "Also check the refresh-token path.", }; + const out = render(args); + const lines = out.split("\n"); - for (const out of [render(args), renderCompleted(args)]) { - expect(out).toContain("Goal"); - expect(out).toContain("Fix rendering."); - expect(out).toContain("Constraints"); - expect(out).toContain("Keep task visible"); - expect(out).not.toContain("# Goal"); - expect(out).not.toContain("# Constraints"); - } + expect(lines[0]).toContain("resume AuthLoader"); + expect(out).toContain("Also check the refresh-token path."); }); // Once the tool produces a result, the container suppresses the call entirely - // via `mergeCallAndResult` and `renderResult` draws each agent. As a safety - // net, `renderCall` also drops its duplicate per-task preview when a result - // snapshot is present, so the two never stack. - it("drops the per-task preview list once a result snapshot exists", () => { + // via `mergeCallAndResult` and `renderResult` draws the agent. As a safety + // net, `renderCall` also drops its preview when a result snapshot is present, + // so the two never stack. + it("drops the preview once a result snapshot exists", () => { const args: TaskParams = { agent: "reviewer", - tasks: [ - { id: "ReviewAuth", description: "Audit the auth module", assignment: "..." }, - { id: "ReviewDb", description: "Audit the db layer", assignment: "..." }, - ], + id: "ReviewAuth", + description: "Audit the auth module", + assignment: "Review the auth module.", }; const component = taskToolRenderer.renderCall( args, @@ -166,7 +122,6 @@ describe("task renderer: streaming call preview", () => { const out = Bun.stripANSI(component.render(160).join("\n")); expect(out).not.toContain("Audit the auth module"); - expect(out).not.toContain("Audit the db layer"); - expect(out).not.toContain("Tasks ("); + expect(out).not.toContain("Review the auth module."); }); }); diff --git a/packages/coding-agent/test/task/render-nested-live.test.ts b/packages/coding-agent/test/task/render-nested-live.test.ts index 422ce07e1..1555476b1 100644 --- a/packages/coding-agent/test/task/render-nested-live.test.ts +++ b/packages/coding-agent/test/task/render-nested-live.test.ts @@ -1,7 +1,7 @@ import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import type { AgentProgress, SingleResult, TaskParams, TaskToolDetails } from "@oh-my-pi/pi-coding-agent/task"; +import type { AgentProgress, SingleResult, TaskToolDetails } from "@oh-my-pi/pi-coding-agent/task"; import { taskToolRenderer } from "@oh-my-pi/pi-coding-agent/task/render"; import { formatDuration, formatNumber } from "@oh-my-pi/pi-utils"; @@ -41,6 +41,7 @@ describe("task renderer: nested live rendering", () => { recentTools: [], recentOutput: [], toolCount: 1, + requests: 0, tokens: 1000, cost: 0, durationMs: 1234, @@ -63,6 +64,7 @@ describe("task renderer: nested live rendering", () => { truncated: false, durationMs: 500, tokens: 200, + requests: 0, }; } @@ -79,6 +81,7 @@ describe("task renderer: nested live rendering", () => { recentTools: [], recentOutput: [], toolCount: 0, + requests: 0, tokens: 0, cost: 0, durationMs: 0, @@ -209,32 +212,6 @@ describe("task renderer: nested live rendering", () => { expect(text).not.toContain("Σ"); }); - it("keeps the shared context visible while the task is in progress", async () => { - const theme = (await getThemeByName("dark"))!; - const details: TaskToolDetails = { - projectAgentsDir: null, - results: [], - totalDurationMs: 0, - progress: [makeRunningProgress({ id: "Probe", description: "Investigate padding" })], - }; - const args = { - agent: "task", - context: "# Goal\nHarden the auth stack before the cut.", - tasks: [], - } as unknown as TaskParams; - const component = taskToolRenderer.renderResult( - { content: [{ type: "text", text: "Running 1 agents..." }], details }, - { expanded: false, isPartial: true, spinnerFrame: 0 }, - theme, - args, - ); - const text = Bun.stripANSI(component.render(160).join("\n")); - // The brief no longer vanishes the moment the first progress snapshot - // replaces the streaming call view. - expect(text).toContain("Goal"); - expect(text).toContain("Harden the auth stack before the cut."); - }); - it("renders a static result header while the body shimmers the running task name", async () => { const theme = (await getThemeByName("dark"))!; const details: TaskToolDetails = { diff --git a/packages/coding-agent/test/task/render-yield-shape.test.ts b/packages/coding-agent/test/task/render-yield-shape.test.ts index f59dd4058..f6f94113d 100644 --- a/packages/coding-agent/test/task/render-yield-shape.test.ts +++ b/packages/coding-agent/test/task/render-yield-shape.test.ts @@ -46,6 +46,7 @@ describe("task renderer: malformed yield slot (#1987)", () => { truncated: false, durationMs: 250, tokens: 100, + requests: 0, // Cast deliberately: production typings declare `unknown[]`, but the // renderer must defend against a stray non-array value — that's // exactly what this regression test exercises. @@ -66,6 +67,7 @@ describe("task renderer: malformed yield slot (#1987)", () => { recentTools: [], recentOutput: [], toolCount: 1, + requests: 0, tokens: 100, cost: 0, durationMs: 250, diff --git a/packages/coding-agent/test/task/subagent-lsp.test.ts b/packages/coding-agent/test/task/subagent-lsp.test.ts index f123cd5e9..42c0edc01 100644 --- a/packages/coding-agent/test/task/subagent-lsp.test.ts +++ b/packages/coding-agent/test/task/subagent-lsp.test.ts @@ -18,7 +18,9 @@ import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus"; const TEST_TASK: TaskParams = { agent: "task", - tasks: [{ id: "CheckLsp", description: "Check LSP availability", assignment: "Inspect LSP tools." }], + id: "CheckLsp", + description: "Check LSP availability", + assignment: "Inspect LSP tools.", }; function createAssistantStopMessage(text: string): AssistantMessage { diff --git a/packages/coding-agent/test/task/task-guards.test.ts b/packages/coding-agent/test/task/task-guards.test.ts new file mode 100644 index 000000000..37134a2ec --- /dev/null +++ b/packages/coding-agent/test/task/task-guards.test.ts @@ -0,0 +1,273 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { LoadExtensionsResult } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; +import type { CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk"; +import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession, AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { formatResultOutputFallback } from "@oh-my-pi/pi-coding-agent/task"; +import { runSubprocess } from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentDefinition } from "@oh-my-pi/pi-coding-agent/task/types"; +import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus"; + +/** + * Contract: runaway-subagent guards. + * + * 1. The executor counts assistant requests (message_end events) and surfaces + * the count on `SingleResult.requests`. + * 2. Crossing the soft request budget injects exactly ONE steering notice into + * the child session asking it to wrap up; crossing 1.5x the budget aborts + * the run gracefully. + * 3. A cancelled/aborted child that produced no completed output salvages its + * last assistant text into a `[cancelled after N req, …]` summary instead + * of the parent seeing "(no output)" and redoing the work. + */ + +interface SteerCall { + content: string; + options?: { deliverAs?: "steer" | "followUp" }; +} + +interface FakeSessionConfig { + /** Events pushed to the executor's subscriber on the next microtask. */ + events?: AgentSessionEvent[]; + /** When true, prompt/waitForIdle hang until abort() is called. */ + hang?: boolean; + /** Returned from getLastAssistantMessage (salvage source). */ + lastAssistantMessage?: unknown; +} + +interface FakeSessionHandle { + session: AgentSession; + steerCalls: SteerCall[]; + abortCalls: () => number; +} + +function assistantMessageEnd(text: string, usage?: Record): AgentSessionEvent { + return { + type: "message_end", + message: { + role: "assistant", + content: text ? [{ type: "text", text }] : [], + usage: usage ?? { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15 }, + }, + } as unknown as AgentSessionEvent; +} + +function yieldToolEnd(): AgentSessionEvent { + return { + type: "tool_execution_end", + toolCallId: "tool-yield", + toolName: "yield", + result: { + content: [{ type: "text", text: "Result submitted." }], + details: { status: "success", data: { ok: true } }, + }, + isError: false, + } as AgentSessionEvent; +} + +function createFakeSession(config: FakeSessionConfig = {}): FakeSessionHandle { + let abortCount = 0; + const steerCalls: SteerCall[] = []; + const { promise: hang, resolve: releaseHang } = Promise.withResolvers(); + if (!config.hang) releaseHang(); + + const session: Partial = { + state: { messages: [] } as never, + agent: { state: { systemPrompt: ["test"] } } as never, + extensionRunner: undefined as never, + sessionManager: { appendSessionInit: () => {} } as never, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async (_names: string[]) => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + if (config.events?.length) { + const events = config.events; + queueMicrotask(() => { + for (const event of events) listener(event); + }); + } + return () => {}; + }, + prompt: async () => { + await hang; + return true; + }, + waitForIdle: async () => { + await hang; + }, + sendUserMessage: async (content, options) => { + steerCalls.push({ content: String(content), options }); + }, + getLastAssistantMessage: () => (config.lastAssistantMessage ?? undefined) as never, + abort: async () => { + abortCount += 1; + releaseHang(); + }, + dispose: async () => {}, + }; + return { + session: session as AgentSession, + steerCalls, + abortCalls: () => abortCount, + }; +} + +function mockCreateAgentSession(session: AgentSession) { + return vi.spyOn(sdkModule, "createAgentSession").mockResolvedValue({ + session, + extensionsResult: {} as unknown as LoadExtensionsResult, + setToolUIContext: () => {}, + eventBus: new EventBus(), + } satisfies CreateAgentSessionResult); +} + +const baseAgent: AgentDefinition = { + name: "task", + description: "test", + systemPrompt: "test", + source: "bundled", +}; + +const baseOptions = { + cwd: "/tmp", + agent: baseAgent, + task: "do work", + index: 0, + id: "subagent-guards", + modelRegistry: { refresh: async () => {} } as unknown as ModelRegistry, + enableLsp: false, +}; + +describe("runSubprocess request guards", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("counts assistant requests into SingleResult.requests", async () => { + const settings = Settings.isolated({ "task.maxRuntimeMs": 0 }); + const handle = createFakeSession({ + events: [ + assistantMessageEnd("step one"), + assistantMessageEnd("step two"), + assistantMessageEnd("step three"), + yieldToolEnd(), + ], + }); + mockCreateAgentSession(handle.session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-requests", settings }); + + expect(result.aborted).toBe(false); + expect(result.requests).toBe(3); + // Well under any budget: no steer injected. + expect(handle.steerCalls.length).toBe(0); + }); + + it("injects exactly one steering notice when the soft budget is crossed", async () => { + // Budget 4: steer fires at request 4 and must not repeat at request 5 + // (still below the 1.5x hard stop of 6). + const settings = Settings.isolated({ "task.maxRuntimeMs": 0, "task.softRequestBudget": 4 }); + const handle = createFakeSession({ + events: [ + assistantMessageEnd("1"), + assistantMessageEnd("2"), + assistantMessageEnd("3"), + assistantMessageEnd("4"), + assistantMessageEnd("5"), + yieldToolEnd(), + ], + }); + mockCreateAgentSession(handle.session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-steer", settings }); + + expect(result.requests).toBe(5); + expect(result.aborted).toBe(false); + expect(handle.steerCalls.length).toBe(1); + expect(handle.steerCalls[0].content).toContain("[budget notice]"); + expect(handle.steerCalls[0].content).toContain("4 requests"); + expect(handle.steerCalls[0].options?.deliverAs).toBe("steer"); + }); + + it("aborts the run gracefully at 1.5x the soft budget", async () => { + // Budget 2: steer at 2, hard stop at 3. The session hangs so only the + // budget abort can release it. + const settings = Settings.isolated({ "task.maxRuntimeMs": 0, "task.softRequestBudget": 2 }); + const handle = createFakeSession({ + hang: true, + events: [ + assistantMessageEnd("", { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15 }), + assistantMessageEnd("", { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15 }), + assistantMessageEnd("", { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15 }), + ], + }); + mockCreateAgentSession(handle.session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-hard-stop", settings }); + + expect(result.aborted).toBe(true); + expect(result.exitCode).toBe(1); + expect(result.abortReason).toContain("request budget exceeded"); + expect(handle.abortCalls()).toBeGreaterThanOrEqual(1); + expect(handle.steerCalls.length).toBe(1); + }); + + it("salvages the last assistant text for an aborted child with no completed output", async () => { + const settings = Settings.isolated({ "task.maxRuntimeMs": 50 }); + const handle = createFakeSession({ + hang: true, + events: [ + // One completed assistant turn with usage but no text content: + // counts a request and tokens without producing output chunks. + assistantMessageEnd("", { input: 100, output: 50, cacheRead: 0, cacheWrite: 0, totalTokens: 150 }), + ], + lastAssistantMessage: { + role: "assistant", + stopReason: "aborted", + content: [{ type: "text", text: "Reading the\n\tconfig loader before patching" }], + }, + }); + mockCreateAgentSession(handle.session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-salvage", settings }); + + expect(result.aborted).toBe(true); + expect(result.requests).toBe(1); + expect(result.output).toContain("cancelled after 1 req"); + expect(result.output).toContain("150 tok"); + expect(result.output).toContain("last activity:"); + // Whitespace is flattened so the snippet stays a single line. + expect(result.output).toContain("Reading the config loader before patching"); + expect(result.output).not.toContain("\n"); + }); + + it("clips oversized salvage snippets", async () => { + const settings = Settings.isolated({ "task.maxRuntimeMs": 50 }); + const longText = `start-marker ${"x".repeat(700)}`; + const handle = createFakeSession({ + hang: true, + lastAssistantMessage: { + role: "assistant", + stopReason: "aborted", + content: [{ type: "text", text: longText }], + }, + }); + mockCreateAgentSession(handle.session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-salvage-clip", settings }); + + expect(result.aborted).toBe(true); + expect(result.output).toContain("start-marker"); + expect(result.output).toContain("…"); + expect(result.output).not.toContain(longText); + expect(result.output.length).toBeLessThan(700); + }); + + it("formats the (no output) fallback with the request count", () => { + expect(formatResultOutputFallback({ output: "", stderr: "", requests: 7 })).toBe("(no output) after 7 req"); + expect(formatResultOutputFallback({ output: " ", stderr: "", requests: 0 })).toBe("(no output)"); + expect(formatResultOutputFallback({ output: "real output", stderr: "", requests: 7 })).toBe("real output"); + expect(formatResultOutputFallback({ output: "", stderr: "boom", requests: 7 })).toBe("boom"); + }); +}); diff --git a/packages/coding-agent/test/task/task-progress-render.test.ts b/packages/coding-agent/test/task/task-progress-render.test.ts index 1a4562582..1465ddca0 100644 --- a/packages/coding-agent/test/task/task-progress-render.test.ts +++ b/packages/coding-agent/test/task/task-progress-render.test.ts @@ -1,9 +1,9 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import type { RenderResultOptions } from "@oh-my-pi/pi-agent-core"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { taskToolRenderer } from "@oh-my-pi/pi-coding-agent/task/render"; -import type { AgentProgress, TaskToolDetails } from "@oh-my-pi/pi-coding-agent/task/types"; +import type { AgentProgress, SingleResult, TaskToolDetails } from "@oh-my-pi/pi-coding-agent/task/types"; function runningProgress(overrides: Partial = {}): AgentProgress { return { @@ -16,6 +16,7 @@ function runningProgress(overrides: Partial = {}): AgentProgress recentTools: [], recentOutput: [], toolCount: 0, + requests: 0, tokens: 0, cost: 0, durationMs: 0, @@ -23,6 +24,24 @@ function runningProgress(overrides: Partial = {}): AgentProgress }; } +function finishedResult(overrides: Partial = {}): SingleResult { + return { + index: 0, + id: "Agent", + agent: "task", + agentSource: "bundled", + task: "investigate hot paths", + exitCode: 0, + output: "done", + stderr: "", + truncated: false, + durationMs: 0, + tokens: 0, + requests: 0, + ...overrides, + }; +} + function detailsFor(progress: AgentProgress): TaskToolDetails { return { projectAgentsDir: null, results: [], totalDurationMs: 0, progress: [progress] }; } @@ -102,7 +121,64 @@ describe("task progress rendering", () => { expect(strippedRow).not.toContain(theme.getSpinnerFrames("status")[0]); }); - it("pins unfinished tasks below finished ones in the live view", async () => { + it("shimmers the pending description like a running one (frozen async spawn snapshot)", async () => { + const theme = (await getThemeByName("dark"))!; + const options: RenderResultOptions = { expanded: false, isPartial: true, spinnerFrame: 0 }; + const progress = runningProgress({ + id: "BestGpt", + status: "pending", + description: "Combine winners for gpt", + }); + + const renderRow = (timeMs: number): string => { + vi.spyOn(Date, "now").mockReturnValue(timeMs); + return findRow( + taskToolRenderer.renderResult( + { content: [{ type: "text", text: "" }], details: detailsFor(progress) }, + options, + theme, + ), + "BestGpt", + ); + }; + + const rawRow0 = renderRow(0); + const rawRow1 = renderRow(700); + + expect(Bun.stripANSI(rawRow0)).toContain("BestGpt: Combine winners for gpt"); + // The label stays one solid bold-accent run; the description shimmers, + // so the row animates across frames exactly like a running agent's. + const label = theme.fg("accent", theme.bold("BestGpt")); + expect(rawRow0).toContain(label); + expect(rawRow1).toContain(label); + expect(rawRow0).not.toBe(rawRow1); + }); + + it("renders the assignment markdown inside the result frame", async () => { + const theme = (await getThemeByName("dark"))!; + setThemeInstance(theme); + const options: RenderResultOptions = { expanded: false, isPartial: true, spinnerFrame: 0 }; + const progress = runningProgress({ id: "BestGpt", status: "pending", description: "Combine winners" }); + + const rendered = Bun.stripANSI( + taskToolRenderer + .renderResult( + { content: [{ type: "text", text: "Spawned agent BestGpt..." }], details: detailsFor(progress) }, + options, + theme, + { agent: "task", id: "BestGpt", assignment: "# Target\nCombine the winning patches." }, + ) + .render(120) + .join("\n"), + ); + + // The brief stays visible for the whole task lifecycle, not just while + // the call args stream in. + expect(rendered).toContain("Target"); + expect(rendered).toContain("Combine the winning patches."); + }); + + it("pins unfinished tasks below finished ones, finished sorted by runtime asc", async () => { const theme = (await getThemeByName("dark"))!; const options: RenderResultOptions = { expanded: false, isPartial: true, spinnerFrame: 0 }; const details: TaskToolDetails = { @@ -110,10 +186,10 @@ describe("task progress rendering", () => { results: [], totalDurationMs: 0, progress: [ - runningProgress({ index: 0, id: "FirstRunning", status: "running" }), - runningProgress({ index: 1, id: "DoneEarly", status: "completed" }), + runningProgress({ index: 0, id: "FirstRunning", status: "running", durationMs: 9000 }), + runningProgress({ index: 1, id: "DoneSlow", status: "completed", durationMs: 5000 }), runningProgress({ index: 2, id: "StillPending", status: "pending" }), - runningProgress({ index: 3, id: "FailedFast", status: "failed" }), + runningProgress({ index: 3, id: "FailedFast", status: "failed", durationMs: 1000 }), ], }; @@ -124,8 +200,34 @@ describe("task progress rendering", () => { .join("\n"), ); - // Finished agents (in dispatch order) come first; pending/running stay at the bottom. - const positions = ["DoneEarly", "FailedFast", "FirstRunning", "StillPending"].map(id => rendered.indexOf(id)); + // Finished agents sorted by runtime ascending; pending/running stay at the + // bottom in dispatch order. + const positions = ["FailedFast", "DoneSlow", "FirstRunning", "StillPending"].map(id => rendered.indexOf(id)); + expect(positions.every(p => p >= 0)).toBe(true); + expect(positions).toEqual([...positions].sort((a, b) => a - b)); + }); + + it("orders finalized results by runtime asc, matching the live view", async () => { + const theme = (await getThemeByName("dark"))!; + const options: RenderResultOptions = { expanded: false, isPartial: false }; + const details: TaskToolDetails = { + projectAgentsDir: null, + results: [ + finishedResult({ index: 0, id: "SlowFinish", durationMs: 9000 }), + finishedResult({ index: 1, id: "FastFinish", durationMs: 1000 }), + finishedResult({ index: 2, id: "MidFinish", durationMs: 4000 }), + ], + totalDurationMs: 9000, + }; + + const rendered = Bun.stripANSI( + taskToolRenderer + .renderResult({ content: [{ type: "text", text: "" }], details }, options, theme) + .render(120) + .join("\n"), + ); + + const positions = ["FastFinish", "MidFinish", "SlowFinish"].map(id => rendered.indexOf(id)); expect(positions.every(p => p >= 0)).toBe(true); expect(positions).toEqual([...positions].sort((a, b) => a - b)); }); @@ -144,15 +246,17 @@ describe("task result detail-less state", () => { it("renders a validation failure with the error glyph, not a success bullet", async () => { const theme = (await getThemeByName("dark"))!; + // The assignment section renders markdown, which reads the active theme. + setThemeInstance(theme); const options: RenderResultOptions = { expanded: false, isPartial: false }; const component = taskToolRenderer.renderResult( { - content: [{ type: "text", text: 'Validation failed for tool "task": tasks: Invalid input' }], + content: [{ type: "text", text: 'Validation failed for tool "task": assignment: Invalid input' }], isError: true, }, options, theme, - { agent: "explore", tasks: [] }, + { agent: "explore", assignment: "Look around." }, ); const stripped = Bun.stripANSI(component.render(120).join("\n")); @@ -166,10 +270,11 @@ describe("task result detail-less state", () => { it("renders a detail-less success with the accent bullet, not an error glyph", async () => { const theme = (await getThemeByName("dark"))!; + setThemeInstance(theme); const options: RenderResultOptions = { expanded: false, isPartial: false }; const component = taskToolRenderer.renderResult({ content: [{ type: "text", text: "done" }] }, options, theme, { agent: "explore", - tasks: [], + assignment: "Look around.", }); const stripped = Bun.stripANSI(component.render(120).join("\n")); diff --git a/packages/coding-agent/test/task/task-resume.test.ts b/packages/coding-agent/test/task/task-resume.test.ts new file mode 100644 index 000000000..3db8d3273 --- /dev/null +++ b/packages/coding-agent/test/task/task-resume.test.ts @@ -0,0 +1,272 @@ +/** + * Contracts: task tool spawn/resume routing (rework-contracts.md §3). + * + * 1. With an AsyncJobManager wired, `execute` returns immediately (agent id + + * job id) while the job body is still gated; job completion delivers a + * result carrying the `task(resume:"")` / `history://` hint. + * 2. Resume routes through `AgentLifecycleManager.ensureLive` and hands the + * live session to `resumeSubprocess`; an ensureLive rejection surfaces as a + * ToolError naming `history://`. + * 3. The session-scoped spawn semaphore (task.maxConcurrency) serializes job + * bodies: with concurrency 1 the second body does not start until the + * first releases. + * + * Param validation (agent XOR resume, resume+isolated, missing assignment) is + * covered by test/task/task-schema.test.ts. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AgentLifecycleManager } from "@oh-my-pi/pi-coding-agent/registry/agent-lifecycle"; +import { AgentRegistry } from "@oh-my-pi/pi-coding-agent/registry/agent-registry"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { TaskTool } from "@oh-my-pi/pi-coding-agent/task"; +import * as discoveryModule from "@oh-my-pi/pi-coding-agent/task/discovery"; +import * as executorModule from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentDefinition, SingleResult, TaskParams } from "@oh-my-pi/pi-coding-agent/task/types"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { ToolError } from "@oh-my-pi/pi-coding-agent/tools/tool-errors"; + +const taskAgent: AgentDefinition = { + name: "task", + description: "General-purpose task agent", + systemPrompt: "You are a task agent.", + source: "bundled", +}; + +function createSession(options: { manager?: AsyncJobManager; settings?: Record }): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated(options.settings ?? {}), + getSessionFile: () => null, + getSessionSpawns: () => "*", + asyncJobManager: options.manager, + } as unknown as ToolSession; +} + +function getFirstText(result: { content: Array<{ type: string; text?: string }> }): string { + const content = result.content.find(part => part.type === "text"); + return content?.type === "text" ? (content.text ?? "") : ""; +} + +function makeResult(id: string, overrides: Partial = {}): SingleResult { + return { + index: 0, + id, + agent: "task", + agentSource: "bundled", + task: "task prompt", + assignment: "Do the thing.", + exitCode: 0, + output: "All done.", + stderr: "", + truncated: false, + durationMs: 5, + tokens: 0, + requests: 1, + ...overrides, + }; +} + +interface Deferred { + promise: Promise; + resolve: () => void; +} + +function deferred(): Deferred { + let resolve!: () => void; + const promise = new Promise(res => { + resolve = res; + }); + return { promise, resolve }; +} + +async function pollUntil(predicate: () => boolean, timeoutMs = 2000): Promise { + const start = Date.now(); + while (!predicate()) { + if (Date.now() - start > timeoutMs) throw new Error("pollUntil timed out"); + await Bun.sleep(5); + } +} + +describe("task spawn/resume routing", () => { + const managers: AsyncJobManager[] = []; + + function createManager(): AsyncJobManager { + const manager = new AsyncJobManager({ onJobComplete: () => {} }); + managers.push(manager); + return manager; + } + + beforeEach(() => { + AgentRegistry.resetGlobalForTests(); + AgentLifecycleManager.resetGlobalForTests(); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + for (const manager of managers.splice(0)) { + await manager.dispose({ timeoutMs: 1000 }); + } + AgentLifecycleManager.resetGlobalForTests(); + AgentRegistry.resetGlobalForTests(); + }); + + it("returns immediately on spawn and delivers the resume hint when the job completes", async () => { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ + agents: [taskAgent], + projectAgentsDir: null, + }); + const gate = deferred(); + const runSpy = vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + await gate.promise; + return makeResult(options.id ?? "?"); + }); + + const manager = createManager(); + const tool = await TaskTool.create(createSession({ manager })); + + const result = await tool.execute("tc-spawn", { + agent: "task", + id: "Spawnling", + description: "background work", + assignment: "Do the thing.", + } as TaskParams); + + // Tool returned while the job body is still gated on the deferred. + const text = getFirstText(result); + expect(text).toContain("Spawned agent `Spawnling`"); + const jobId = result.details?.async?.jobId; + expect(jobId).toBeTruthy(); + expect(text).toContain(`job \`${jobId}\``); + const job = manager.getJob(jobId!); + expect(job?.status).toBe("running"); + expect(job?.resultText).toBeUndefined(); + + gate.resolve(); + await job!.promise; + + expect(job!.status).toBe("completed"); + expect(job!.resultText).toContain('task(resume:"Spawnling")'); + expect(job!.resultText).toContain("history://Spawnling"); + expect(runSpy).toHaveBeenCalledTimes(1); + }); + + it("rejects an async resume of an unregistered agent without registering a job", async () => { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ + agents: [taskAgent], + projectAgentsDir: null, + }); + const manager = createManager(); + const tool = await TaskTool.create(createSession({ manager })); + + const error = await tool + .execute("tc-resume-unknown", { resume: "Nobody", assignment: "Follow up." } as TaskParams) + .then( + () => null, + err => err as Error, + ); + + expect(error).toBeInstanceOf(ToolError); + expect(error?.message).toContain('Unknown agent "Nobody"'); + expect(error?.message).toContain("history://Nobody"); + expect(manager.getAllJobs()).toHaveLength(0); + }); + + it("resume routes through AgentLifecycleManager.ensureLive and hands the live session to resumeSubprocess", async () => { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [], projectAgentsDir: null }); + const fakeSession = { messages: [] } as unknown as AgentSession; + AgentRegistry.global().register({ + id: "Reso", + displayName: "task", + kind: "sub", + session: fakeSession, + status: "idle", + }); + const ensureLiveSpy = vi.spyOn(AgentLifecycleManager.global(), "ensureLive").mockResolvedValue(fakeSession); + const resumeSpy = vi + .spyOn(executorModule, "resumeSubprocess") + .mockResolvedValue(makeResult("Reso", { output: "Follow-up done." })); + + // No job manager => sync fallback, so the resume pipeline runs inline. + const tool = await TaskTool.create(createSession({})); + const result = await tool.execute("tc-resume", { + resume: "Reso", + assignment: "Also check refresh tokens.", + } as TaskParams); + + expect(ensureLiveSpy).toHaveBeenCalledTimes(1); + expect(ensureLiveSpy).toHaveBeenCalledWith("Reso"); + expect(resumeSpy).toHaveBeenCalledTimes(1); + const resumeOptions = resumeSpy.mock.calls[0]![0]; + expect(resumeOptions.session).toBe(fakeSession); + expect(resumeOptions.id).toBe("Reso"); + expect(resumeOptions.assignment).toBe("Also check refresh tokens."); + + const text = getFirstText(result); + expect(text).toContain("Reso"); + expect(text).toContain("completed"); + expect(result.details?.results).toHaveLength(1); + expect(result.details?.results[0]?.exitCode).toBe(0); + }); + + it("surfaces an ensureLive rejection as a ToolError naming history://", async () => { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [], projectAgentsDir: null }); + vi.spyOn(AgentLifecycleManager.global(), "ensureLive").mockRejectedValue(new Error("session file corrupt")); + + const tool = await TaskTool.create(createSession({})); + const error = await tool + .execute("tc-resume-dead", { resume: "Ghost", assignment: "Wake up." } as TaskParams) + .then( + () => null, + err => err as Error, + ); + + expect(error).toBeInstanceOf(ToolError); + expect(error?.message).toContain('Cannot resume "Ghost"'); + expect(error?.message).toContain("session file corrupt"); + expect(error?.message).toContain("history://Ghost"); + }); + + it("bounds concurrent job bodies with the session spawn semaphore", async () => { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ + agents: [taskAgent], + projectAgentsDir: null, + }); + const started: string[] = []; + const gates = new Map(); + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async options => { + const id = options.id ?? "?"; + started.push(id); + const gate = deferred(); + gates.set(id, gate); + await gate.promise; + return makeResult(id); + }); + + const manager = createManager(); + const tool = await TaskTool.create(createSession({ manager, settings: { "task.maxConcurrency": 1 } })); + + const first = await tool.execute("tc-1", { agent: "task", id: "First", assignment: "Work A." } as TaskParams); + const second = await tool.execute("tc-2", { agent: "task", id: "Second", assignment: "Work B." } as TaskParams); + const firstJob = manager.getJob(first.details!.async!.jobId)!; + const secondJob = manager.getJob(second.details!.async!.jobId)!; + + // First job body reaches the executor; second stays parked at the semaphore. + await pollUntil(() => started.length >= 1); + await Bun.sleep(25); + expect(started).toHaveLength(1); + + // Releasing the first body lets the second one start. + gates.get(started[0]!)!.resolve(); + await firstJob.promise; + await pollUntil(() => started.length === 2); + expect(started).toEqual(["First", "Second"]); + + gates.get("Second")!.resolve(); + await secondJob.promise; + expect(firstJob.status).toBe("completed"); + expect(secondJob.status).toBe("completed"); + }); +}); diff --git a/packages/coding-agent/test/task/task-schema.test.ts b/packages/coding-agent/test/task/task-schema.test.ts new file mode 100644 index 000000000..0e3f6ecb7 --- /dev/null +++ b/packages/coding-agent/test/task/task-schema.test.ts @@ -0,0 +1,83 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { TaskTool, taskSchema } from "@oh-my-pi/pi-coding-agent/task"; +import * as discoveryModule from "@oh-my-pi/pi-coding-agent/task/discovery"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; + +// Contract (rework-contracts.md §3): the task tool spawns ONE agent per call. +// `tasks[]` and `context` are gone; `resume` continues an existing agent. + +describe("task schema (single-spawn)", () => { + it("accepts {agent, assignment}", () => { + const parsed = taskSchema.safeParse({ agent: "explore", assignment: "Map the auth module." }); + expect(parsed.success).toBe(true); + }); + + it("accepts {resume, assignment}", () => { + const parsed = taskSchema.safeParse({ resume: "AuthLoader", assignment: "Also check refresh tokens." }); + expect(parsed.success).toBe(true); + }); + + it("requires assignment", () => { + const parsed = taskSchema.safeParse({ agent: "explore" }); + expect(parsed.success).toBe(false); + }); + + it("carries no tasks/context fields", () => { + const parsed = taskSchema.safeParse({ + agent: "explore", + assignment: "Map the auth module.", + context: "shared background", + tasks: [{ id: "A", assignment: "..." }], + }); + expect(parsed.success).toBe(true); + if (parsed.success) { + // Unknown keys are stripped: the batch/context shape no longer exists. + expect("tasks" in parsed.data).toBe(false); + expect("context" in parsed.data).toBe(false); + } + }); +}); + +describe("task spawn/resume validation", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + function createSession(): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated({ "task.isolation.mode": "none" }), + getSessionFile: () => null, + getSessionSpawns: () => "*", + } as unknown as ToolSession; + } + + async function executeText(params: unknown): Promise { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [], projectAgentsDir: null }); + const tool = await TaskTool.create(createSession()); + const result = await tool.execute("tool-call", params); + return result.content.find(part => part.type === "text")?.text ?? ""; + } + + it("rejects resume + agent together", async () => { + const text = await executeText({ agent: "explore", resume: "AuthLoader", assignment: "..." }); + expect(text).toContain("not both"); + }); + + it("rejects neither resume nor agent", async () => { + const text = await executeText({ assignment: "..." }); + expect(text).toContain("Missing `agent`"); + }); + + it("rejects resume + isolated", async () => { + const text = await executeText({ resume: "AuthLoader", isolated: true, assignment: "..." }); + expect(text).toContain("not resumable"); + }); + + it("rejects a missing assignment", async () => { + const text = await executeText({ agent: "explore" }); + expect(text).toContain("Missing `assignment`"); + }); +}); diff --git a/packages/coding-agent/test/tool-live-region-scrollback.test.ts b/packages/coding-agent/test/tool-live-region-scrollback.test.ts index d94b5691b..7434134b0 100644 --- a/packages/coding-agent/test/tool-live-region-scrollback.test.ts +++ b/packages/coding-agent/test/tool-live-region-scrollback.test.ts @@ -1,6 +1,6 @@ import { beforeAll, describe, expect, it } from "bun:test"; import type { AssistantMessage } from "@oh-my-pi/pi-ai"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { Settings, settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message"; import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; import { TranscriptContainer } from "@oh-my-pi/pi-coding-agent/modes/components/transcript-container"; @@ -563,19 +563,21 @@ describe("tool live-region scrollback", () => { } }); - it("commits the scrolled-off head of an over-tall pending task context to scrollback", async () => { + it("commits the scrolled-off head of an over-tall pending eval cell to scrollback", async () => { if (process.platform === "win32") return; + // The single-spawn task renderer bounds its pending preview (the old + // uncapped multi-task `context` field is gone), so the eval tool — + // whose pending code preview is intentionally never capped — now + // carries the over-tall pending content. const term = new VirtualTerminal(120, 12); const tui = new TUI(term); const chat = new TranscriptContainer(); - const context = (n: number) => Array.from({ length: n }, (_unused, i) => `- CTX-${i}`).join("\n"); + const code = (n: number) => Array.from({ length: n }, (_unused, i) => `// - CTX-${i}`).join("\n"); const args = (n: number) => ({ - agent: "task", - context: context(n), - tasks: [{ id: "alpha", description: "probe", assignment: "Inspect the task context." }], + cells: [{ language: "js", title: "probe", code: code(n) }], }); - const component = new ToolExecutionComponent("task", args(4), {}, undefined, tui, process.cwd()); + const component = new ToolExecutionComponent("eval", args(4), {}, undefined, tui, process.cwd()); try { chat.addChild(component); @@ -603,35 +605,39 @@ describe("tool live-region scrollback", () => { } }); - it("keeps the static task context reachable in scrollback while progress ticks below it", async () => { + it("keeps the static task assignment reachable in scrollback while progress ticks below it", async () => { if (process.platform === "win32") return; const term = new VirtualTerminal(120, 12); const tui = new TUI(term); const chat = new TranscriptContainer(); - const context = Array.from({ length: 40 }, (_unused, i) => `- CTX-${i}`).join("\n"); - const args = { - agent: "explore", - context, - tasks: [{ id: "alpha", description: "probe", assignment: "Inspect the repo." }], - }; + const assignment = Array.from({ length: 40 }, (_unused, i) => `- CTX-${i}`).join("\n"); + const args = { agent: "explore", id: "alpha", description: "probe", assignment }; const component = new ToolExecutionComponent("task", args, {}, undefined, tui, process.cwd()); - const progressAt = (toolCount: number) => ({ + // The multi-line assignment section only renders expanded; shimmer + // would repaint the status line above it every frame, capping the + // stable prefix above the assignment, so pin it off for the run. + component.setExpanded(true); + settings.override("display.shimmer", "disabled"); + const progressAt = (tick: number) => ({ index: 0, id: "alpha", agent: "explore", agentSource: "bundled" as const, status: "running" as const, - task: "probe", + task: assignment, description: "probe", + currentTool: "read", + currentToolArgs: `probe-step-${tick}`, recentTools: [], recentOutput: [], - toolCount, + toolCount: 5, + requests: 0, tokens: 0, cost: 0, - durationMs: toolCount * 250, + durationMs: 1000, }); - const partial = (toolCount: number) => + const partial = (tick: number) => component.updateResult( { content: [{ type: "text", text: "" }], @@ -639,7 +645,7 @@ describe("tool live-region scrollback", () => { projectAgentsDir: null, results: [], totalDurationMs: 0, - progress: [progressAt(toolCount)], + progress: [progressAt(tick)], }, }, true, @@ -651,13 +657,14 @@ describe("tool live-region scrollback", () => { tui.start(); await term.waitForRender(); - // A running task rewrites its progress line (tool counts, spinner) - // below the static context for the whole run. The context head that - // scrolled above the viewport must still reach native scrollback — - // previously the ticking tail suspended commits for the entire - // block, leaving the context neither in history nor on screen. - // Two full promotion windows: the call→result transition frame - // poisons the first window's minimum, the second promotes the head. + // A running task rewrites its current-tool line (the ticking tail) + // below the static assignment section for the whole run. The + // assignment head that scrolled above the viewport must still reach + // native scrollback — previously the ticking tail suspended commits + // for the entire block, leaving the assignment neither in history + // nor on screen. Two full promotion windows: the call→result + // transition frame poisons the first window's minimum, the second + // promotes the head. for (let i = 1; i <= 70; i++) { partial(i); tui.requestRender(); @@ -669,8 +676,9 @@ describe("tool live-region scrollback", () => { expect(viewportText).not.toContain("CTX-0"); expect(scrollText).toContain("CTX-0"); - expect(scrollText).toContain("CTX-20"); + expect(scrollText).toContain("CTX-5"); } finally { + settings.clearOverride("display.shimmer"); component.stopAnimation(); tui.stop(); await term.flush(); diff --git a/packages/coding-agent/test/tools/task-repair-args.test.ts b/packages/coding-agent/test/tools/task-repair-args.test.ts index dec3c57cd..4629b327d 100644 --- a/packages/coding-agent/test/tools/task-repair-args.test.ts +++ b/packages/coding-agent/test/tools/task-repair-args.test.ts @@ -44,31 +44,27 @@ describe("repairDoubleEncodedJsonString", () => { }); describe("repairTaskParams", () => { - it("repairs context and each task's assignment/description, leaving ids intact", () => { + it("repairs assignment and description, leaving agent/id intact", () => { const params = { agent: "task", - context: "# Goal\\nDo the thing \\u2014 carefully", - tasks: [ - { - id: "FirstTask", - description: 'judge \\"sketch\\" accuracy', - assignment: "Score 0-100.\\nUse the full range.\\nNo bunching.", - }, - ], + id: "FirstTask", + description: 'judge \\"sketch\\" accuracy', + assignment: "Score 0-100.\\nUse the full range.\\nNo bunching.", } as unknown as TaskParams; const repaired = repairTaskParams(params); - expect(repaired.context).toBe("# Goal\nDo the thing — carefully"); - expect(repaired.tasks[0].id).toBe("FirstTask"); - expect(repaired.tasks[0].description).toBe('judge "sketch" accuracy'); - expect(repaired.tasks[0].assignment).toBe("Score 0-100.\nUse the full range.\nNo bunching."); + expect(repaired.agent).toBe("task"); + expect(repaired.id).toBe("FirstTask"); + expect(repaired.description).toBe('judge "sketch" accuracy'); + expect(repaired.assignment).toBe("Score 0-100.\nUse the full range.\nNo bunching."); }); it("returns the same reference when nothing needs repair", () => { const params = { agent: "task", - context: "plain context", - tasks: [{ id: "A", description: "label", assignment: "do work" }], + id: "A", + description: "label", + assignment: "do work", } as unknown as TaskParams; expect(repairTaskParams(params)).toBe(params); }); diff --git a/packages/coding-agent/test/tools/task-simple-mode.test.ts b/packages/coding-agent/test/tools/task-simple-mode.test.ts index 1e7636e87..3a1189ef3 100644 --- a/packages/coding-agent/test/tools/task-simple-mode.test.ts +++ b/packages/coding-agent/test/tools/task-simple-mode.test.ts @@ -16,6 +16,8 @@ const TEST_AGENTS = [ }, ]; +const ALL_MODES = ["default", "schema-free", "independent"] as const; + function createSession(overrides: Partial> = {}): ToolSession { return { cwd: "/tmp", @@ -36,87 +38,86 @@ function getFirstText(result: { content: Array<{ type: string; text?: string }> return content?.type === "text" ? (content.text ?? "") : ""; } +function mockDiscovery(): void { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ + agents: TEST_AGENTS, + projectAgentsDir: null, + }); +} + describe("task.simple", () => { afterEach(() => { vi.restoreAllMocks(); }); - it("removes only the custom schema input in schema-free mode", async () => { - vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ - agents: TEST_AGENTS, - projectAgentsDir: null, - }); + it("exposes the custom schema input only in default mode", async () => { + mockDiscovery(); - const tool = await TaskTool.create(createSession({ "task.simple": "schema-free" })); - const properties = getSchemaProperties(tool); + const defaultTool = await TaskTool.create(createSession({ "task.simple": "default" })); + expect(getSchemaProperties(defaultTool).schema).toBeDefined(); + expect(defaultTool.description).toContain("- `schema`:"); - expect(properties.context).toBeDefined(); - expect(properties.schema).toBeUndefined(); - expect(tool.description).toContain("`context` or `assignment`"); - expect(tool.description).toContain("- `context`:"); - expect(tool.description).not.toContain("- `schema`:"); + for (const mode of ["schema-free", "independent"] as const) { + const tool = await TaskTool.create(createSession({ "task.simple": mode })); + expect(getSchemaProperties(tool).schema).toBeUndefined(); + expect(tool.description).not.toContain("- `schema`:"); + } }); - it("removes both context and schema inputs in independent mode", async () => { - vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ - agents: TEST_AGENTS, - projectAgentsDir: null, - }); + it("never exposes batch tasks or shared context inputs in any mode", async () => { + mockDiscovery(); - const tool = await TaskTool.create(createSession({ "task.simple": "independent" })); - const properties = getSchemaProperties(tool); - - expect(properties.context).toBeUndefined(); - expect(properties.schema).toBeUndefined(); - expect(tool.description).toContain("each `assignment`"); - expect(tool.description).not.toContain("- `context`:"); - expect(tool.description).not.toContain("- `schema`:"); + for (const mode of ALL_MODES) { + const tool = await TaskTool.create(createSession({ "task.simple": mode })); + const properties = getSchemaProperties(tool); + expect(properties.tasks).toBeUndefined(); + expect(properties.context).toBeUndefined(); + // The flat single-spawn contract is what replaced them. + expect(properties.assignment).toBeDefined(); + expect(properties.resume).toBeDefined(); + } }); - it("rejects direct schema and context fields when the mode disables them", async () => { - vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ - agents: TEST_AGENTS, - projectAgentsDir: null, - }); + it("describes the non-blocking spawn and resume contract", async () => { + mockDiscovery(); - const schemaFreeTool = await TaskTool.create(createSession({ "task.simple": "schema-free" })); - const schemaFreeResult = await schemaFreeTool.execute("tool-1", { - agent: "task", - schema: '{"properties":{"ok":{"type":"boolean"}}}', - tasks: [{ id: "One", description: "label", assignment: "Do the thing." }], - } as TaskParams); - expect(getFirstText(schemaFreeResult)).toContain("does not accept `schema`"); - const validatedSchemaFreeParams = validateToolArguments(schemaFreeTool, { - type: "toolCall", - id: "tool-1-validated", - name: schemaFreeTool.name, - arguments: { + const tool = await TaskTool.create(createSession({ "task.simple": "default" })); + expect(tool.description).toContain("Spawning is non-blocking"); + expect(tool.description).toContain("revives an idle/parked agent"); + }); + + it("rejects a direct schema input when the mode disables it", async () => { + mockDiscovery(); + + for (const mode of ["schema-free", "independent"] as const) { + const tool = await TaskTool.create(createSession({ "task.simple": mode })); + + // Execution-time guard: raw params carrying `schema` are refused. + const result = await tool.execute(`tool-${mode}`, { agent: "task", + id: "One", + description: "label", + assignment: "Do the thing.", schema: '{"properties":{"ok":{"type":"boolean"}}}', - tasks: [{ id: "One", description: "label", assignment: "Do the thing." }], - }, - }); - const validatedSchemaFreeResult = await schemaFreeTool.execute("tool-1-validated", validatedSchemaFreeParams); - expect(getFirstText(validatedSchemaFreeResult)).toContain("does not accept `schema`"); + } as TaskParams); + expect(getFirstText(result)).toContain("does not accept `schema`"); - const independentTool = await TaskTool.create(createSession({ "task.simple": "independent" })); - const independentResult = await independentTool.execute("tool-2", { - agent: "task", - context: "Shared background", - tasks: [{ id: "Two", description: "label", assignment: "Do the independent thing." }], - } as TaskParams); - expect(getFirstText(independentResult)).toContain("does not accept `context`"); - const validatedIndependentParams = validateToolArguments(independentTool, { - type: "toolCall", - id: "tool-2-validated", - name: independentTool.name, - arguments: { - agent: "task", - context: "Shared background", - tasks: [{ id: "Two", description: "label", assignment: "Do the independent thing." }], - }, - }); - const validatedIndependentResult = await independentTool.execute("tool-2-validated", validatedIndependentParams); - expect(getFirstText(validatedIndependentResult)).toContain("does not accept `context`"); + // Round-trip guard: wire validation passes the extraneous `schema` + // through, so the execution-time check must still refuse it. + const validated = validateToolArguments(tool, { + type: "toolCall", + id: `tool-${mode}-validated`, + name: tool.name, + arguments: { + agent: "task", + id: "One", + description: "label", + assignment: "Do the thing.", + schema: '{"properties":{"ok":{"type":"boolean"}}}', + }, + }) as TaskParams; + const validatedResult = await tool.execute(`tool-${mode}-validated`, validated); + expect(getFirstText(validatedResult)).toContain("does not accept `schema`"); + } }); }); diff --git a/packages/swarm-extension/src/swarm/pipeline.ts b/packages/swarm-extension/src/swarm/pipeline.ts index 82c247c14..d2e05464d 100644 --- a/packages/swarm-extension/src/swarm/pipeline.ts +++ b/packages/swarm-extension/src/swarm/pipeline.ts @@ -185,6 +185,7 @@ export class PipelineController { truncated: false, durationMs: 0, tokens: 0, + requests: 0, error, }; return { agentName, result: failResult };