fix(prewalk): stopped completion turn loop

Limited the hidden continuation safety net to the assistant turn immediately following the plan nudge, so later bash-only completion ends normally.

Added regression coverage for commit-style flows that never call edit or write.

Fixes #5551
This commit is contained in:
roboomp
2026-07-15 08:15:28 +00:00
parent 485b31e8f1
commit 2fe124987b
3 changed files with 68 additions and 17 deletions
+1
View File
@@ -13,6 +13,7 @@
### Fixed
- Fixed Bash internal URLs remaining unresolved when used as unquoted arguments inside command substitutions ([#5535](https://github.com/can1357/oh-my-pi/issues/5535)).
- Fixed prewalk repeatedly continuing after a bash-only task such as `commit` had already completed ([#5551](https://github.com/can1357/oh-my-pi/issues/5551)).
## [16.5.2] - 2026-07-14
@@ -1748,6 +1748,8 @@ export class AgentSession {
#prewalk: Prewalk | undefined;
/** True once the plan nudge has been queued; scrubbed from context at the switch. */
#prewalkPlanInjected = false;
/** True until the first assistant turn after the plan nudge completes. */
#prewalkContinuePending = false;
/** True once any successful `todo` call landed — opens the prewalk
* trigger gate: the switch fires at the first edit/write AFTER the todo
* list exists (sessions without an ACTIVE todo tool skip the gate). */
@@ -2247,23 +2249,22 @@ export class AgentSession {
const prewalk = this.#prewalk;
if (!prewalk || context?.message.role !== "assistant") return;
// Structural safety net: every branch below assumes the agent loop will
// run another turn. It won't if THIS turn had no tool calls — the loop
// treats a text-only turn as "the agent is done" and ends the session
// with no further prompting. The plan nudge explicitly asks for a prose
// reply, which makes a text-only turn common right after it — observed
// silently killing production SWE-bench runs before any code was ever
// written. Force one more turn only in that specific, self-created
// hazard window.
if (this.#prewalkPlanInjected && context.toolResults.length === 0) {
this.agent.steer({
role: "custom",
customType: PREWALK_CONTINUE_MESSAGE_TYPE,
content: prewalkContinuePrompt,
attribution: "agent",
display: false,
timestamp: Date.now(),
});
// The plan nudge can produce a prose-only reply, which the agent loop
// treats as completion before any implementation starts. Keep the
// safety net open only for the turn immediately following that nudge:
// once the model calls any tool, later text-only completion is genuine.
if (this.#prewalkContinuePending) {
this.#prewalkContinuePending = false;
if (context.toolResults.length === 0) {
this.agent.steer({
role: "custom",
customType: PREWALK_CONTINUE_MESSAGE_TYPE,
content: prewalkContinuePrompt,
attribution: "agent",
display: false,
timestamp: Date.now(),
});
}
}
// Todo gate: the plan nudge instructs "finish the plan, then init the
@@ -2284,6 +2285,7 @@ export class AgentSession {
if (!action) {
if (!this.#prewalkPlanInjected) {
this.#prewalkPlanInjected = true;
this.#prewalkContinuePending = true;
this.agent.steer({
role: "custom",
customType: PREWALK_PLAN_MESSAGE_TYPE,
@@ -2344,6 +2346,7 @@ export class AgentSession {
}
this.#prewalk = { target, thinkingLevel };
this.#prewalkPlanInjected = true;
this.#prewalkContinuePending = true;
this.agent.steer({
role: "custom",
customType: PREWALK_PLAN_MESSAGE_TYPE,
@@ -291,6 +291,53 @@ describe("AgentSession prewalk", () => {
]);
expect(session.model?.id).toBe(target.id);
});
it("does not continue a completed bash-only task after the plan-nudge window closes", async () => {
const primary = modelOrThrow("claude-sonnet-4-5");
const target = modelOrThrow("claude-sonnet-4-6");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const mock = createMockModel({
responses: [
toolCall("t1", "record"),
toolCall("t2", "bash"),
{ content: [{ type: "text", text: "Commit complete." }], stopReason: "stop" },
],
});
const requested: string[] = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
model: primary,
systemPrompt: ["Test"],
tools: [recordTool as AgentTool, bashTool as AgentTool],
messages: [],
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (model, context, options) => {
requested.push(`${model.provider}/${model.id}`);
return mock.stream(model, context, options);
},
});
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated({ "compaction.enabled": false }),
modelRegistry,
toolRegistry,
prewalk: { target },
});
await session.prompt("commit the current changes");
expect(requested).toEqual([
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
]);
});
it("skips the todo gate when todo is registered but not active (subagent-style restricted slates)", async () => {
// Regression: the gate used to key on the tool REGISTRY, so a session
// whose active-tool slate excluded `todo` (subagents strip it) while the