fix(session): added #prunedTerminalRefusal field to store the

- Added `#prunedTerminalRefusal` field to store the pruned refusal for post-settle consumers.
- Modified `getLastAssistantMessage()` to return the pruned refusal before active-context lookup.
- Reset `#prunedTerminalRefusal` on `agent_start` so a fresh run supersedes the settled refusal.
- Updated test mock helpers to include `getLastAssistantMessage` for consistency.
This commit is contained in:
can1357
2026-07-18 17:51:48 +02:00
parent 127cca511a
commit c655db4e3c
4 changed files with 80 additions and 2 deletions
@@ -4446,6 +4446,11 @@ export class AgentSession {
}
#processAgentEvent = async (event: AgentEvent): Promise<void> => {
// A fresh run supersedes the previously settled (and pruned) refusal
// turn: state-based lookups take over again.
if (event.type === "agent_start") {
this.#prunedTerminalRefusal = undefined;
}
// Step the mid-run todo counter synchronously, BEFORE any await in this
// handler. The agent loop's next-turn `getAsideMessages` poll can run
// before queued microtasks drain, so `#takeMidRunTodoNudge` MUST see the
@@ -4991,10 +4996,14 @@ export class AgentSession {
}
// Classifier refusals are persisted-skipped above; also prune the trailing
// stub from active context so the next turn's prompt does not replay it.
// Keep a reference for post-settle readers (print mode, task executor via
// getLastAssistantMessage) — pruning made the terminal error invisible to
// anything inspecting agent state after prompt() resolved.
// Fall through to the standard error tail so `session_stop` hooks (block,
// continue, telemetry) still fire — matching the pre-fix flow for
// `stopReason === "error"`.
if (this.#isClassifierRefusal(msg)) {
this.#prunedTerminalRefusal = msg;
this.#removeAssistantMessageFromActiveContext(msg);
}
this.#resolveRetry();
@@ -6962,9 +6971,14 @@ export class AgentSession {
}
}
/** Most recent assistant message in agent state. */
/**
* Most recent settled assistant message. A classifier-refusal turn pruned
* from active context at settle is still reported until the next run
* starts, so terminal-outcome consumers (print mode, task executor) see
* the refusal error rather than the previous turn — or nothing.
*/
getLastAssistantMessage(): AssistantMessage | undefined {
return this.#findLastAssistantMessage();
return this.#prunedTerminalRefusal ?? this.#findLastAssistantMessage();
}
/** Current effective system prompt blocks (includes any per-turn extension modifications) */
get systemPrompt(): string[] {
@@ -1119,6 +1119,68 @@ describe("AgentSession retry fallback", () => {
});
});
it("keeps the pruned refusal visible to getLastAssistantMessage until the next run", async () => {
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!primaryModel) {
throw new Error("Expected bundled test model to exist");
}
const mock = createMockModel({
responses: [
{
stopReason: "error",
stopDetails: { type: "refusal", category: "cyber", explanation: "Declined." },
errorMessage: "Refusal (cyber): Declined.",
},
{ content: ["recovered"] },
],
});
const agent = new Agent({
getApiKey: model => `${model.provider}-test-key`,
initialState: {
model: primaryModel,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: (model, context, options) => mock.stream(model, context, options),
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.baseDelayMs": 5,
"retry.maxRetries": 1,
"retry.modelFallback": false,
});
settings.setModelRole("default", `${primaryModel.provider}/${primaryModel.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
await session.prompt("Trigger classifier refusal");
await session.waitForIdle();
// The refusal turn is pruned from active context (no assistant tail)…
expect(session.agent.state.messages.at(-1)?.role).toBe("user");
// …but terminal-outcome consumers (print mode, task executor) must still
// see the settled error instead of a silently successful-looking state.
const settled = session.getLastAssistantMessage();
expect(settled?.stopReason).toBe("error");
expect(settled?.errorMessage).toBe("Refusal (cyber): Declined.");
expect(settled?.stopDetails).toEqual({ type: "refusal", category: "cyber", explanation: "Declined." });
await session.prompt("Next prompt supersedes the pruned refusal");
await session.waitForIdle();
const recovered = session.getLastAssistantMessage();
expect(recovered?.stopReason).toBe("stop");
expect(recovered?.content).toEqual([{ type: "text", text: "recovered" }]);
});
it("does not exceed retry.maxRetries for classifier fallback chains", async () => {
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
const firstFallback = getBundledModel("openai", "gpt-4o-mini");
@@ -49,6 +49,7 @@ function createDelayedSession(finalMessage: AssistantMessage): DelayedSession {
messages.push(finalMessage);
return true;
},
getLastAssistantMessage: () => messages.findLast(message => message.role === "assistant"),
dispose: async () => {},
} as unknown as AgentSession;
@@ -50,6 +50,7 @@ function createMockSession(
extensionRunner: undefined,
subscribe: () => () => {},
prompt: async () => {},
getLastAssistantMessage: () => messages.findLast(message => message.role === "assistant"),
dispose,
} as unknown as AgentSession;
}