fix(agent): excluded reasoning from empty-stop diagnosis

This commit is contained in:
can1357
2026-08-16 02:30:32 +02:00
parent 14813330f5
commit 44e50af907
3 changed files with 43 additions and 6 deletions
+1 -1
View File
@@ -102,7 +102,7 @@
- Fixed Ctrl+G external editors failing to launch on Windows because Bun re-quoted the embedded `cmd.exe /c` command line ([#8544](https://github.com/can1357/oh-my-pi/issues/8544)).
### Fixed
- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with `usage.output > 0` means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only stops keep a thinking block (and bill output for it), so they retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)).
- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with billed output beyond any provider-reported reasoning usage means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only and known reasoning-only stops retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)).
## [17.3.3] - 2026-08-14
@@ -678,16 +678,20 @@ export class TurnRecovery {
if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) {
const attempts = this.#emptyStopRetryCount - 1;
const outputTokens = assistantMessage.usage.output;
const outputTokensExcludingKnownReasoning = Math.max(
0,
outputTokens - (assistantMessage.usage.reasoningTokens ?? 0),
);
let finalError: string;
if (providerEmptyOutput) {
finalError = "Assistant returned no final output after retry cap; try switching models";
} else if (outputTokens > 0 && assistantMessage.content.length === 0) {
// Billed output on a truly zero-block stop means content was generated and
// then dropped downstream (a filter/refusal flattened to
} else if (outputTokensExcludingKnownReasoning > 0 && assistantMessage.content.length === 0) {
// Billed non-reasoning output on a truly zero-block stop means content was
// generated and then dropped downstream (a filter/refusal flattened to
// `finish_reason: "stop"` by a proxy, or a lossy API translation) — the
// context/`/shake images` hint is wrong here, so name the billed output
// instead. Thinking-only stops keep a thinking block (and bill output for
// it), so they fall through to the context hint rather than this path.
// instead. Known reasoning-only usage is not evidence that deliverable
// content was dropped, and thinking-only stops retain a thinking block.
finalError = `Assistant returned an empty stop after retry cap, but the provider billed ${outputTokens} output token${outputTokens === 1 ? "" : "s"} for it; content was generated and then dropped before delivery, which usually points to a provider-side content filter or a lossy API translation rather than a context problem`;
} else {
finalError =
@@ -72,6 +72,14 @@ function filteredEmptyStop(): MockResponse {
};
}
function reasoningOnlyEmptyStop(): MockResponse {
return {
content: [],
stopReason: "stop",
usage: { output: 126, reasoningTokens: 126, cacheRead: 100 },
};
}
function orphanedToolUseStop(): MockResponse {
return {
content: [{ type: "thinking", thinking: "I should call a tool next." }],
@@ -499,6 +507,31 @@ describe("AgentSession empty stop guard", () => {
expect(finalError).not.toContain("/shake images");
});
it("keeps the context hint when a capped zero-block stop billed only reasoning tokens", async () => {
const { session, mock } = await createHarness([
reasoningOnlyEmptyStop(),
reasoningOnlyEmptyStop(),
reasoningOnlyEmptyStop(),
reasoningOnlyEmptyStop(),
]);
const retryEndEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_end" }>> = [];
session.subscribe(event => {
if (event.type === "auto_retry_end") {
retryEndEvents.push(event);
}
});
await expectPromptCompletes(session.prompt("think without delivering an answer"));
await session.waitForIdle();
expect(mock.calls).toHaveLength(4);
expect(retryEndEvents).toHaveLength(1);
expect(retryEndEvents[0]?.success).toBe(false);
const finalError = retryEndEvents[0]?.finalError ?? "";
expect(finalError).toContain("/shake images");
expect(finalError).not.toContain("billed");
});
it("keeps the context hint for a capped thinking-only stop even though it billed output", async () => {
const { session, mock } = await createHarness([
thinkingOnlyStop(),