From cefa91494f37d5f35e0f398c8da81e8d6e2b8cd9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 08:28:14 +0000 Subject: [PATCH 1/3] fix(prewalk): keep continuation armed across todo turns The continuation net now stays armed across a todo-only turn and disarms only when a non-planning tool runs without a prose plan, so the normal plan-nudge to todo to prose to edit flow still reaches implementation while a bash-only completion no longer loops. Fixes #5551 --- .../coding-agent/src/session/agent-session.ts | 25 +++++--- .../test/agent-session-prewalk.test.ts | 59 +++++++++++++++++++ 2 files changed, 76 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 298511762..bff982bba 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2249,13 +2249,23 @@ export class AgentSession { const prewalk = this.#prewalk; if (!prewalk || context?.message.role !== "assistant") return; - // The plan nudge can produce a prose-only reply, which the agent loop - // treats as completion before any implementation starts. Keep the - // safety net open only for the turn immediately following that nudge: - // once the model calls any tool, later text-only completion is genuine. + const todoCalledThisTurn = context.toolResults.some(result => result.toolName === "todo"); + if (todoCalledThisTurn) { + this.#prewalkTodoSeen = true; + } + + // The plan nudge asks for a prose plan (optionally alongside the todo + // init) before implementation begins; the agent loop would treat that + // text-only reply as terminal and end the run with no code written, so + // the continuation net forces one more turn. It stays armed across a + // todo-only turn — the plan is captured but the prose follow-up and the + // first edit/write are still to come — and disarms the moment the model + // takes any other action without a prose plan, so a task that finishes + // with a prose reply after only non-planning tools (e.g. a bash-only + // commit) is never forced to loop. if (this.#prewalkContinuePending) { - this.#prewalkContinuePending = false; if (context.toolResults.length === 0) { + this.#prewalkContinuePending = false; this.agent.steer({ role: "custom", customType: PREWALK_CONTINUE_MESSAGE_TYPE, @@ -2264,6 +2274,8 @@ export class AgentSession { display: false, timestamp: Date.now(), }); + } else if (!todoCalledThisTurn) { + this.#prewalkContinuePending = false; } } @@ -2275,9 +2287,6 @@ export class AgentSession { // ACTIVE tool set, not the registry: a registered-but-deactivated todo // (e.g. a restricted active-tool slate) is uncallable and would // deadlock the switch. - if (context.toolResults.some(result => result.toolName === "todo")) { - this.#prewalkTodoSeen = true; - } const todoGateOpen = this.#prewalkTodoSeen || !this.getActiveToolNames().includes("todo"); const action = todoGateOpen ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) diff --git a/packages/coding-agent/test/agent-session-prewalk.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts index a4eaee53a..8bdb2a246 100644 --- a/packages/coding-agent/test/agent-session-prewalk.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -338,6 +338,65 @@ describe("AgentSession prewalk", () => { ]); }); + it("keeps the continuation net armed across a todo-only turn so a following prose reply still implements", async () => { + // Regression: the plan nudge asks for a prose plan plus the todo init. + // A model that answers the nudge with a todo-only turn, then a prose + // follow-up, must not end the run before any edit/write — the todo turn + // is part of planning, not completion, so the continuation net stays + // armed until an action tool runs. + const primary = modelOrThrow("claude-sonnet-4-5"); + const target = modelOrThrow("claude-sonnet-4-6"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + + // Turn 1: read-only (nudge injected after). Turn 2: todo — plan + // captured, gate opens, net stays armed. Turn 3: prose — must be + // bridged, not treated as terminal. Turn 4: write — switch. + const mock = createMockModel({ + responses: [ + toolCall("t1", "record"), + toolCall("t2", "todo"), + { content: [{ type: "text", text: "Plan captured, starting now." }], stopReason: "stop" }, + toolCall("t4", "write"), + { content: ["done"] }, + ], + }); + const requested: string[] = []; + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { + model: primary, + systemPrompt: ["Test"], + tools: [recordTool as AgentTool, writeTool as AgentTool, todoTool as AgentTool], + messages: [], + thinkingLevel: Effort.Medium, + }, + convertToLlm, + streamFn: (model, context, options) => { + requested.push(`${model.provider}/${model.id}`); + return mock.stream(model, context, options); + }, + }); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry, + toolRegistry, + prewalk: { target }, + }); + + await session.prompt("do the task"); + + expect(requested).toEqual([ + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${target.provider}/${target.id}`, + ]); + expect(session.model?.id).toBe(target.id); + }); + it("skips the todo gate when todo is registered but not active (subagent-style restricted slates)", async () => { // Regression: the gate used to key on the tool REGISTRY, so a session // whose active-tool slate excluded `todo` (subagents strip it) while the From b993c0d10c421d9951d521d7b498c28e3b07030e Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 08:38:00 +0000 Subject: [PATCH 2/3] fix(prewalk): make continuation net one-shot A prose plan and a bash-only completion are structurally identical after any pre-implementation tool detour, so the net now stays armed across every such turn and fires exactly once on the first text-only reply. This bridges read/record/bash detours before the plan while bounding a genuine no-edit completion to a single continuation instead of looping. Fixes #5551 --- .../coding-agent/src/session/agent-session.ts | 41 +++++++++---------- .../test/agent-session-prewalk.test.ts | 15 ++++++- 2 files changed, 33 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index bff982bba..d7e2617f1 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2255,28 +2255,25 @@ export class AgentSession { } // The plan nudge asks for a prose plan (optionally alongside the todo - // init) before implementation begins; the agent loop would treat that - // text-only reply as terminal and end the run with no code written, so - // the continuation net forces one more turn. It stays armed across a - // todo-only turn — the plan is captured but the prose follow-up and the - // first edit/write are still to come — and disarms the moment the model - // takes any other action without a prose plan, so a task that finishes - // with a prose reply after only non-planning tools (e.g. a bash-only - // commit) is never forced to loop. - if (this.#prewalkContinuePending) { - if (context.toolResults.length === 0) { - this.#prewalkContinuePending = false; - this.agent.steer({ - role: "custom", - customType: PREWALK_CONTINUE_MESSAGE_TYPE, - content: prewalkContinuePrompt, - attribution: "agent", - display: false, - timestamp: Date.now(), - }); - } else if (!todoCalledThisTurn) { - this.#prewalkContinuePending = false; - } + // init) before implementation begins. The agent loop treats a text-only + // reply as terminal, so without help the run ends before any code is + // written. The continuation net forces exactly one more turn: it stays + // armed across any number of pre-implementation tool turns (exploratory + // read/record/bash and the todo init alike) and fires on the first + // text-only reply, whenever it arrives. Firing at most once bounds the + // hazard: a task that genuinely finishes without an edit/write (e.g. a + // bash-only commit) gets a single "continue" nudge and then ends when it + // replies text-only again, rather than looping forever (#5551). + if (this.#prewalkContinuePending && context.toolResults.length === 0) { + this.#prewalkContinuePending = false; + this.agent.steer({ + role: "custom", + customType: PREWALK_CONTINUE_MESSAGE_TYPE, + content: prewalkContinuePrompt, + attribution: "agent", + display: false, + timestamp: Date.now(), + }); } // Todo gate: the plan nudge instructs "finish the plan, then init the diff --git a/packages/coding-agent/test/agent-session-prewalk.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts index 8bdb2a246..9644f555e 100644 --- a/packages/coding-agent/test/agent-session-prewalk.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -292,16 +292,25 @@ describe("AgentSession prewalk", () => { expect(session.model?.id).toBe(target.id); }); - it("does not continue a completed bash-only task after the plan-nudge window closes", async () => { + it("bounds a completed bash-only task to a single continuation instead of looping", async () => { + // Regression (#5551): with no edit/write ever run, the continuation net + // used to re-fire on every text-only reply, looping forever. It must + // fire at most once — one "continue" nudge — then let the next text-only + // reply end the run. No mock fallback: a stray extra turn rejects. const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + // Turn 1: record (nudge injected after). Turn 2: bash — not an action + // tool. Turn 3: prose — the single continuation fires. Turn 4: prose + // again — no more continuation, run ends. A 5th call would exhaust the + // script and reject. const mock = createMockModel({ responses: [ toolCall("t1", "record"), toolCall("t2", "bash"), { content: [{ type: "text", text: "Commit complete." }], stopReason: "stop" }, + { content: [{ type: "text", text: "Nothing left to do." }], stopReason: "stop" }, ], }); const requested: string[] = []; @@ -331,11 +340,15 @@ describe("AgentSession prewalk", () => { await session.prompt("commit the current changes"); + // Exactly one continuation: 4 turns, all on the primary (no edit/write, + // so no switch), then a clean stop. expect(requested).toEqual([ `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, ]); + expect(session.model?.id).toBe(primary.id); }); it("keeps the continuation net armed across a todo-only turn so a following prose reply still implements", async () => { From 7b07ad7d6a6d5e5fde8c9dc12efad7714209b117 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 08:44:54 +0000 Subject: [PATCH 3/3] fix(prewalk): rearmed continuation after tool progress Each tool-result turn now re-arms one text-only continuation while prewalk is pending. Consecutive prose replies without intervening tool progress terminate naturally, preserving multi-step plan detours without restoring the completion loop. Fixes #5551 --- .../coding-agent/src/session/agent-session.ts | 24 +++++++++---------- .../test/agent-session-prewalk.test.ts | 23 +++++++++--------- 2 files changed, 24 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index d7e2617f1..d44123031 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1748,7 +1748,7 @@ export class AgentSession { #prewalk: Prewalk | undefined; /** True once the plan nudge has been queued; scrubbed from context at the switch. */ #prewalkPlanInjected = false; - /** True until the first assistant turn after the plan nudge completes. */ + /** Armed by plan/tool progress; consumed by one text-only continuation. */ #prewalkContinuePending = false; /** True once any successful `todo` call landed — opens the prewalk * trigger gate: the switch fires at the first edit/write AFTER the todo @@ -2254,17 +2254,17 @@ export class AgentSession { this.#prewalkTodoSeen = true; } - // The plan nudge asks for a prose plan (optionally alongside the todo - // init) before implementation begins. The agent loop treats a text-only - // reply as terminal, so without help the run ends before any code is - // written. The continuation net forces exactly one more turn: it stays - // armed across any number of pre-implementation tool turns (exploratory - // read/record/bash and the todo init alike) and fires on the first - // text-only reply, whenever it arrives. Firing at most once bounds the - // hazard: a task that genuinely finishes without an edit/write (e.g. a - // bash-only commit) gets a single "continue" nudge and then ends when it - // replies text-only again, rather than looping forever (#5551). - if (this.#prewalkContinuePending && context.toolResults.length === 0) { + // The plan nudge asks for a prose plan before implementation begins, + // but the agent loop treats each text-only reply as terminal. Tool + // progress re-arms one continuation, allowing split flows such as + // plan → todo → prose → read → prose → edit/write. Consuming the arm + // before steering also detects completion: two consecutive text-only + // replies have no intervening progress, so the second ends naturally + // instead of producing the #5551 loop. + const hasToolResults = context.toolResults.length > 0; + if (this.#prewalkPlanInjected && hasToolResults) { + this.#prewalkContinuePending = true; + } else if (this.#prewalkContinuePending) { this.#prewalkContinuePending = false; this.agent.steer({ role: "custom", diff --git a/packages/coding-agent/test/agent-session-prewalk.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts index 9644f555e..e981091de 100644 --- a/packages/coding-agent/test/agent-session-prewalk.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -351,25 +351,25 @@ describe("AgentSession prewalk", () => { expect(session.model?.id).toBe(primary.id); }); - it("keeps the continuation net armed across a todo-only turn so a following prose reply still implements", async () => { - // Regression: the plan nudge asks for a prose plan plus the todo init. - // A model that answers the nudge with a todo-only turn, then a prose - // follow-up, must not end the run before any edit/write — the todo turn - // is part of planning, not completion, so the continuation net stays - // armed until an action tool runs. + it("re-arms continuation after tool progress between prose turns", async () => { + // Regression: a normal prewalk can split planning across several turns: + // prose plan, todo init, then prose before implementation. Each tool + // progress segment must earn one continuation so the second prose turn + // cannot end the run before edit/write. const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - // Turn 1: read-only (nudge injected after). Turn 2: todo — plan - // captured, gate opens, net stays armed. Turn 3: prose — must be - // bridged, not treated as terminal. Turn 4: write — switch. + // Turn 1: read-only (nudge injected after). Turn 2: prose plan — + // bridged. Turn 3: todo — gate opens and re-arms the net. Turn 4: + // prose — bridged again. Turn 5: write — switch. const mock = createMockModel({ responses: [ toolCall("t1", "record"), - toolCall("t2", "todo"), + { content: [{ type: "text", text: "Here is the plan." }], stopReason: "stop" }, + toolCall("t3", "todo"), { content: [{ type: "text", text: "Plan captured, starting now." }], stopReason: "stop" }, - toolCall("t4", "write"), + toolCall("t5", "write"), { content: ["done"] }, ], }); @@ -405,6 +405,7 @@ describe("AgentSession prewalk", () => { `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, `${target.provider}/${target.id}`, ]); expect(session.model?.id).toBe(target.id);