fix(prewalk): stopped completion turn loop
Limited the hidden continuation safety net to the assistant turn immediately following the plan nudge, so later bash-only completion ends normally. Added regression coverage for commit-style flows that never call edit or write. Fixes #5551
This commit is contained in:
@@ -13,6 +13,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Bash internal URLs remaining unresolved when used as unquoted arguments inside command substitutions ([#5535](https://github.com/can1357/oh-my-pi/issues/5535)).
|
||||
- Fixed prewalk repeatedly continuing after a bash-only task such as `commit` had already completed ([#5551](https://github.com/can1357/oh-my-pi/issues/5551)).
|
||||
|
||||
## [16.5.2] - 2026-07-14
|
||||
|
||||
|
||||
@@ -1748,6 +1748,8 @@ export class AgentSession {
|
||||
#prewalk: Prewalk | undefined;
|
||||
/** True once the plan nudge has been queued; scrubbed from context at the switch. */
|
||||
#prewalkPlanInjected = false;
|
||||
/** True until the first assistant turn after the plan nudge completes. */
|
||||
#prewalkContinuePending = false;
|
||||
/** True once any successful `todo` call landed — opens the prewalk
|
||||
* trigger gate: the switch fires at the first edit/write AFTER the todo
|
||||
* list exists (sessions without an ACTIVE todo tool skip the gate). */
|
||||
@@ -2247,23 +2249,22 @@ export class AgentSession {
|
||||
const prewalk = this.#prewalk;
|
||||
if (!prewalk || context?.message.role !== "assistant") return;
|
||||
|
||||
// Structural safety net: every branch below assumes the agent loop will
|
||||
// run another turn. It won't if THIS turn had no tool calls — the loop
|
||||
// treats a text-only turn as "the agent is done" and ends the session
|
||||
// with no further prompting. The plan nudge explicitly asks for a prose
|
||||
// reply, which makes a text-only turn common right after it — observed
|
||||
// silently killing production SWE-bench runs before any code was ever
|
||||
// written. Force one more turn only in that specific, self-created
|
||||
// hazard window.
|
||||
if (this.#prewalkPlanInjected && context.toolResults.length === 0) {
|
||||
this.agent.steer({
|
||||
role: "custom",
|
||||
customType: PREWALK_CONTINUE_MESSAGE_TYPE,
|
||||
content: prewalkContinuePrompt,
|
||||
attribution: "agent",
|
||||
display: false,
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
// The plan nudge can produce a prose-only reply, which the agent loop
|
||||
// treats as completion before any implementation starts. Keep the
|
||||
// safety net open only for the turn immediately following that nudge:
|
||||
// once the model calls any tool, later text-only completion is genuine.
|
||||
if (this.#prewalkContinuePending) {
|
||||
this.#prewalkContinuePending = false;
|
||||
if (context.toolResults.length === 0) {
|
||||
this.agent.steer({
|
||||
role: "custom",
|
||||
customType: PREWALK_CONTINUE_MESSAGE_TYPE,
|
||||
content: prewalkContinuePrompt,
|
||||
attribution: "agent",
|
||||
display: false,
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Todo gate: the plan nudge instructs "finish the plan, then init the
|
||||
@@ -2284,6 +2285,7 @@ export class AgentSession {
|
||||
if (!action) {
|
||||
if (!this.#prewalkPlanInjected) {
|
||||
this.#prewalkPlanInjected = true;
|
||||
this.#prewalkContinuePending = true;
|
||||
this.agent.steer({
|
||||
role: "custom",
|
||||
customType: PREWALK_PLAN_MESSAGE_TYPE,
|
||||
@@ -2344,6 +2346,7 @@ export class AgentSession {
|
||||
}
|
||||
this.#prewalk = { target, thinkingLevel };
|
||||
this.#prewalkPlanInjected = true;
|
||||
this.#prewalkContinuePending = true;
|
||||
this.agent.steer({
|
||||
role: "custom",
|
||||
customType: PREWALK_PLAN_MESSAGE_TYPE,
|
||||
|
||||
@@ -291,6 +291,53 @@ describe("AgentSession prewalk", () => {
|
||||
]);
|
||||
expect(session.model?.id).toBe(target.id);
|
||||
});
|
||||
|
||||
it("does not continue a completed bash-only task after the plan-nudge window closes", async () => {
|
||||
const primary = modelOrThrow("claude-sonnet-4-5");
|
||||
const target = modelOrThrow("claude-sonnet-4-6");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
|
||||
const mock = createMockModel({
|
||||
responses: [
|
||||
toolCall("t1", "record"),
|
||||
toolCall("t2", "bash"),
|
||||
{ content: [{ type: "text", text: "Commit complete." }], stopReason: "stop" },
|
||||
],
|
||||
});
|
||||
const requested: string[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
model: primary,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [recordTool as AgentTool, bashTool as AgentTool],
|
||||
messages: [],
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (model, context, options) => {
|
||||
requested.push(`${model.provider}/${model.id}`);
|
||||
return mock.stream(model, context, options);
|
||||
},
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry,
|
||||
toolRegistry,
|
||||
prewalk: { target },
|
||||
});
|
||||
|
||||
await session.prompt("commit the current changes");
|
||||
|
||||
expect(requested).toEqual([
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
]);
|
||||
});
|
||||
|
||||
it("skips the todo gate when todo is registered but not active (subagent-style restricted slates)", async () => {
|
||||
// Regression: the gate used to key on the tool REGISTRY, so a session
|
||||
// whose active-tool slate excluded `todo` (subagents strip it) while the
|
||||
|
||||
Reference in New Issue
Block a user