diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b8ac93088..e059aed75 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -102,7 +102,7 @@ - Fixed Ctrl+G external editors failing to launch on Windows because Bun re-quoted the embedded `cmd.exe /c` command line ([#8544](https://github.com/can1357/oh-my-pi/issues/8544)). ### Fixed -- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with `usage.output > 0` means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only stops keep a thinking block (and bill output for it), so they retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)). +- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with billed output beyond any provider-reported reasoning usage means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only and known reasoning-only stops retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)). ## [17.3.3] - 2026-08-14 diff --git a/packages/coding-agent/src/session/turn-recovery.ts b/packages/coding-agent/src/session/turn-recovery.ts index 1e8a692cc..79b57adad 100644 --- a/packages/coding-agent/src/session/turn-recovery.ts +++ b/packages/coding-agent/src/session/turn-recovery.ts @@ -678,16 +678,20 @@ export class TurnRecovery { if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) { const attempts = this.#emptyStopRetryCount - 1; const outputTokens = assistantMessage.usage.output; + const outputTokensExcludingKnownReasoning = Math.max( + 0, + outputTokens - (assistantMessage.usage.reasoningTokens ?? 0), + ); let finalError: string; if (providerEmptyOutput) { finalError = "Assistant returned no final output after retry cap; try switching models"; - } else if (outputTokens > 0 && assistantMessage.content.length === 0) { - // Billed output on a truly zero-block stop means content was generated and - // then dropped downstream (a filter/refusal flattened to + } else if (outputTokensExcludingKnownReasoning > 0 && assistantMessage.content.length === 0) { + // Billed non-reasoning output on a truly zero-block stop means content was + // generated and then dropped downstream (a filter/refusal flattened to // `finish_reason: "stop"` by a proxy, or a lossy API translation) — the // context/`/shake images` hint is wrong here, so name the billed output - // instead. Thinking-only stops keep a thinking block (and bill output for - // it), so they fall through to the context hint rather than this path. + // instead. Known reasoning-only usage is not evidence that deliverable + // content was dropped, and thinking-only stops retain a thinking block. finalError = `Assistant returned an empty stop after retry cap, but the provider billed ${outputTokens} output token${outputTokens === 1 ? "" : "s"} for it; content was generated and then dropped before delivery, which usually points to a provider-side content filter or a lossy API translation rather than a context problem`; } else { finalError = diff --git a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts index ef1b51e0d..3078ab7f5 100644 --- a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts +++ b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts @@ -72,6 +72,14 @@ function filteredEmptyStop(): MockResponse { }; } +function reasoningOnlyEmptyStop(): MockResponse { + return { + content: [], + stopReason: "stop", + usage: { output: 126, reasoningTokens: 126, cacheRead: 100 }, + }; +} + function orphanedToolUseStop(): MockResponse { return { content: [{ type: "thinking", thinking: "I should call a tool next." }], @@ -499,6 +507,31 @@ describe("AgentSession empty stop guard", () => { expect(finalError).not.toContain("/shake images"); }); + it("keeps the context hint when a capped zero-block stop billed only reasoning tokens", async () => { + const { session, mock } = await createHarness([ + reasoningOnlyEmptyStop(), + reasoningOnlyEmptyStop(), + reasoningOnlyEmptyStop(), + reasoningOnlyEmptyStop(), + ]); + const retryEndEvents: Array> = []; + session.subscribe(event => { + if (event.type === "auto_retry_end") { + retryEndEvents.push(event); + } + }); + + await expectPromptCompletes(session.prompt("think without delivering an answer")); + await session.waitForIdle(); + + expect(mock.calls).toHaveLength(4); + expect(retryEndEvents).toHaveLength(1); + expect(retryEndEvents[0]?.success).toBe(false); + const finalError = retryEndEvents[0]?.finalError ?? ""; + expect(finalError).toContain("/shake images"); + expect(finalError).not.toContain("billed"); + }); + it("keeps the context hint for a capped thinking-only stop even though it billed output", async () => { const { session, mock } = await createHarness([ thinkingOnlyStop(),