fix(prewalk): keep continuation armed across todo turns
The continuation net now stays armed across a todo-only turn and disarms only when a non-planning tool runs without a prose plan, so the normal plan-nudge to todo to prose to edit flow still reaches implementation while a bash-only completion no longer loops. Fixes #5551
This commit is contained in:
@@ -2249,13 +2249,23 @@ export class AgentSession {
|
||||
const prewalk = this.#prewalk;
|
||||
if (!prewalk || context?.message.role !== "assistant") return;
|
||||
|
||||
// The plan nudge can produce a prose-only reply, which the agent loop
|
||||
// treats as completion before any implementation starts. Keep the
|
||||
// safety net open only for the turn immediately following that nudge:
|
||||
// once the model calls any tool, later text-only completion is genuine.
|
||||
const todoCalledThisTurn = context.toolResults.some(result => result.toolName === "todo");
|
||||
if (todoCalledThisTurn) {
|
||||
this.#prewalkTodoSeen = true;
|
||||
}
|
||||
|
||||
// The plan nudge asks for a prose plan (optionally alongside the todo
|
||||
// init) before implementation begins; the agent loop would treat that
|
||||
// text-only reply as terminal and end the run with no code written, so
|
||||
// the continuation net forces one more turn. It stays armed across a
|
||||
// todo-only turn — the plan is captured but the prose follow-up and the
|
||||
// first edit/write are still to come — and disarms the moment the model
|
||||
// takes any other action without a prose plan, so a task that finishes
|
||||
// with a prose reply after only non-planning tools (e.g. a bash-only
|
||||
// commit) is never forced to loop.
|
||||
if (this.#prewalkContinuePending) {
|
||||
this.#prewalkContinuePending = false;
|
||||
if (context.toolResults.length === 0) {
|
||||
this.#prewalkContinuePending = false;
|
||||
this.agent.steer({
|
||||
role: "custom",
|
||||
customType: PREWALK_CONTINUE_MESSAGE_TYPE,
|
||||
@@ -2264,6 +2274,8 @@ export class AgentSession {
|
||||
display: false,
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
} else if (!todoCalledThisTurn) {
|
||||
this.#prewalkContinuePending = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2275,9 +2287,6 @@ export class AgentSession {
|
||||
// ACTIVE tool set, not the registry: a registered-but-deactivated todo
|
||||
// (e.g. a restricted active-tool slate) is uncallable and would
|
||||
// deadlock the switch.
|
||||
if (context.toolResults.some(result => result.toolName === "todo")) {
|
||||
this.#prewalkTodoSeen = true;
|
||||
}
|
||||
const todoGateOpen = this.#prewalkTodoSeen || !this.getActiveToolNames().includes("todo");
|
||||
const action = todoGateOpen
|
||||
? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName])
|
||||
|
||||
@@ -338,6 +338,65 @@ describe("AgentSession prewalk", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps the continuation net armed across a todo-only turn so a following prose reply still implements", async () => {
|
||||
// Regression: the plan nudge asks for a prose plan plus the todo init.
|
||||
// A model that answers the nudge with a todo-only turn, then a prose
|
||||
// follow-up, must not end the run before any edit/write — the todo turn
|
||||
// is part of planning, not completion, so the continuation net stays
|
||||
// armed until an action tool runs.
|
||||
const primary = modelOrThrow("claude-sonnet-4-5");
|
||||
const target = modelOrThrow("claude-sonnet-4-6");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
|
||||
// Turn 1: read-only (nudge injected after). Turn 2: todo — plan
|
||||
// captured, gate opens, net stays armed. Turn 3: prose — must be
|
||||
// bridged, not treated as terminal. Turn 4: write — switch.
|
||||
const mock = createMockModel({
|
||||
responses: [
|
||||
toolCall("t1", "record"),
|
||||
toolCall("t2", "todo"),
|
||||
{ content: [{ type: "text", text: "Plan captured, starting now." }], stopReason: "stop" },
|
||||
toolCall("t4", "write"),
|
||||
{ content: ["done"] },
|
||||
],
|
||||
});
|
||||
const requested: string[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
model: primary,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [recordTool as AgentTool, writeTool as AgentTool, todoTool as AgentTool],
|
||||
messages: [],
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (model, context, options) => {
|
||||
requested.push(`${model.provider}/${model.id}`);
|
||||
return mock.stream(model, context, options);
|
||||
},
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry,
|
||||
toolRegistry,
|
||||
prewalk: { target },
|
||||
});
|
||||
|
||||
await session.prompt("do the task");
|
||||
|
||||
expect(requested).toEqual([
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${target.provider}/${target.id}`,
|
||||
]);
|
||||
expect(session.model?.id).toBe(target.id);
|
||||
});
|
||||
|
||||
it("skips the todo gate when todo is registered but not active (subagent-style restricted slates)", async () => {
|
||||
// Regression: the gate used to key on the tool REGISTRY, so a session
|
||||
// whose active-tool slate excluded `todo` (subagents strip it) while the
|
||||
|
||||
Reference in New Issue
Block a user