From cefa91494f37d5f35e0f398c8da81e8d6e2b8cd9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 08:28:14 +0000 Subject: [PATCH] fix(prewalk): keep continuation armed across todo turns The continuation net now stays armed across a todo-only turn and disarms only when a non-planning tool runs without a prose plan, so the normal plan-nudge to todo to prose to edit flow still reaches implementation while a bash-only completion no longer loops. Fixes #5551 --- .../coding-agent/src/session/agent-session.ts | 25 +++++--- .../test/agent-session-prewalk.test.ts | 59 +++++++++++++++++++ 2 files changed, 76 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 298511762..bff982bba 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2249,13 +2249,23 @@ export class AgentSession { const prewalk = this.#prewalk; if (!prewalk || context?.message.role !== "assistant") return; - // The plan nudge can produce a prose-only reply, which the agent loop - // treats as completion before any implementation starts. Keep the - // safety net open only for the turn immediately following that nudge: - // once the model calls any tool, later text-only completion is genuine. + const todoCalledThisTurn = context.toolResults.some(result => result.toolName === "todo"); + if (todoCalledThisTurn) { + this.#prewalkTodoSeen = true; + } + + // The plan nudge asks for a prose plan (optionally alongside the todo + // init) before implementation begins; the agent loop would treat that + // text-only reply as terminal and end the run with no code written, so + // the continuation net forces one more turn. It stays armed across a + // todo-only turn — the plan is captured but the prose follow-up and the + // first edit/write are still to come — and disarms the moment the model + // takes any other action without a prose plan, so a task that finishes + // with a prose reply after only non-planning tools (e.g. a bash-only + // commit) is never forced to loop. if (this.#prewalkContinuePending) { - this.#prewalkContinuePending = false; if (context.toolResults.length === 0) { + this.#prewalkContinuePending = false; this.agent.steer({ role: "custom", customType: PREWALK_CONTINUE_MESSAGE_TYPE, @@ -2264,6 +2274,8 @@ export class AgentSession { display: false, timestamp: Date.now(), }); + } else if (!todoCalledThisTurn) { + this.#prewalkContinuePending = false; } } @@ -2275,9 +2287,6 @@ export class AgentSession { // ACTIVE tool set, not the registry: a registered-but-deactivated todo // (e.g. a restricted active-tool slate) is uncallable and would // deadlock the switch. - if (context.toolResults.some(result => result.toolName === "todo")) { - this.#prewalkTodoSeen = true; - } const todoGateOpen = this.#prewalkTodoSeen || !this.getActiveToolNames().includes("todo"); const action = todoGateOpen ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) diff --git a/packages/coding-agent/test/agent-session-prewalk.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts index a4eaee53a..8bdb2a246 100644 --- a/packages/coding-agent/test/agent-session-prewalk.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -338,6 +338,65 @@ describe("AgentSession prewalk", () => { ]); }); + it("keeps the continuation net armed across a todo-only turn so a following prose reply still implements", async () => { + // Regression: the plan nudge asks for a prose plan plus the todo init. + // A model that answers the nudge with a todo-only turn, then a prose + // follow-up, must not end the run before any edit/write — the todo turn + // is part of planning, not completion, so the continuation net stays + // armed until an action tool runs. + const primary = modelOrThrow("claude-sonnet-4-5"); + const target = modelOrThrow("claude-sonnet-4-6"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + + // Turn 1: read-only (nudge injected after). Turn 2: todo — plan + // captured, gate opens, net stays armed. Turn 3: prose — must be + // bridged, not treated as terminal. Turn 4: write — switch. + const mock = createMockModel({ + responses: [ + toolCall("t1", "record"), + toolCall("t2", "todo"), + { content: [{ type: "text", text: "Plan captured, starting now." }], stopReason: "stop" }, + toolCall("t4", "write"), + { content: ["done"] }, + ], + }); + const requested: string[] = []; + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { + model: primary, + systemPrompt: ["Test"], + tools: [recordTool as AgentTool, writeTool as AgentTool, todoTool as AgentTool], + messages: [], + thinkingLevel: Effort.Medium, + }, + convertToLlm, + streamFn: (model, context, options) => { + requested.push(`${model.provider}/${model.id}`); + return mock.stream(model, context, options); + }, + }); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry, + toolRegistry, + prewalk: { target }, + }); + + await session.prompt("do the task"); + + expect(requested).toEqual([ + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${primary.provider}/${primary.id}`, + `${target.provider}/${target.id}`, + ]); + expect(session.model?.id).toBe(target.id); + }); + it("skips the todo gate when todo is registered but not active (subagent-style restricted slates)", async () => { // Regression: the gate used to key on the tool REGISTRY, so a session // whose active-tool slate excluded `todo` (subagents strip it) while the