fix(agent): excluded reasoning from empty-stop diagnosis
This commit is contained in:
@@ -102,7 +102,7 @@
|
||||
- Fixed Ctrl+G external editors failing to launch on Windows because Bun re-quoted the embedded `cmd.exe /c` command line ([#8544](https://github.com/can1357/oh-my-pi/issues/8544)).
|
||||
### Fixed
|
||||
|
||||
- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with `usage.output > 0` means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only stops keep a thinking block (and bill output for it), so they retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)).
|
||||
- Fixed the capped empty-stop failure always naming the context/`/shake images` hint even when the provider billed output tokens. A zero-block `stop` (no content blocks) with billed output beyond any provider-reported reasoning usage means content was generated and dropped downstream (a filter/refusal flattened to `finish_reason: "stop"` by a proxy, or a lossy API translation), so the message now reports the billed output-token count and points at a provider-side filter/translation instead of a context problem, and logs `outputTokens` alongside the existing warning fields. Thinking-only and known reasoning-only stops retain the context hint ([#8511](https://github.com/can1357/oh-my-pi/issues/8511)).
|
||||
|
||||
## [17.3.3] - 2026-08-14
|
||||
|
||||
|
||||
@@ -678,16 +678,20 @@ export class TurnRecovery {
|
||||
if (this.#emptyStopRetryCount > EMPTY_STOP_MAX_RETRIES) {
|
||||
const attempts = this.#emptyStopRetryCount - 1;
|
||||
const outputTokens = assistantMessage.usage.output;
|
||||
const outputTokensExcludingKnownReasoning = Math.max(
|
||||
0,
|
||||
outputTokens - (assistantMessage.usage.reasoningTokens ?? 0),
|
||||
);
|
||||
let finalError: string;
|
||||
if (providerEmptyOutput) {
|
||||
finalError = "Assistant returned no final output after retry cap; try switching models";
|
||||
} else if (outputTokens > 0 && assistantMessage.content.length === 0) {
|
||||
// Billed output on a truly zero-block stop means content was generated and
|
||||
// then dropped downstream (a filter/refusal flattened to
|
||||
} else if (outputTokensExcludingKnownReasoning > 0 && assistantMessage.content.length === 0) {
|
||||
// Billed non-reasoning output on a truly zero-block stop means content was
|
||||
// generated and then dropped downstream (a filter/refusal flattened to
|
||||
// `finish_reason: "stop"` by a proxy, or a lossy API translation) — the
|
||||
// context/`/shake images` hint is wrong here, so name the billed output
|
||||
// instead. Thinking-only stops keep a thinking block (and bill output for
|
||||
// it), so they fall through to the context hint rather than this path.
|
||||
// instead. Known reasoning-only usage is not evidence that deliverable
|
||||
// content was dropped, and thinking-only stops retain a thinking block.
|
||||
finalError = `Assistant returned an empty stop after retry cap, but the provider billed ${outputTokens} output token${outputTokens === 1 ? "" : "s"} for it; content was generated and then dropped before delivery, which usually points to a provider-side content filter or a lossy API translation rather than a context problem`;
|
||||
} else {
|
||||
finalError =
|
||||
|
||||
@@ -72,6 +72,14 @@ function filteredEmptyStop(): MockResponse {
|
||||
};
|
||||
}
|
||||
|
||||
function reasoningOnlyEmptyStop(): MockResponse {
|
||||
return {
|
||||
content: [],
|
||||
stopReason: "stop",
|
||||
usage: { output: 126, reasoningTokens: 126, cacheRead: 100 },
|
||||
};
|
||||
}
|
||||
|
||||
function orphanedToolUseStop(): MockResponse {
|
||||
return {
|
||||
content: [{ type: "thinking", thinking: "I should call a tool next." }],
|
||||
@@ -499,6 +507,31 @@ describe("AgentSession empty stop guard", () => {
|
||||
expect(finalError).not.toContain("/shake images");
|
||||
});
|
||||
|
||||
it("keeps the context hint when a capped zero-block stop billed only reasoning tokens", async () => {
|
||||
const { session, mock } = await createHarness([
|
||||
reasoningOnlyEmptyStop(),
|
||||
reasoningOnlyEmptyStop(),
|
||||
reasoningOnlyEmptyStop(),
|
||||
reasoningOnlyEmptyStop(),
|
||||
]);
|
||||
const retryEndEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_end" }>> = [];
|
||||
session.subscribe(event => {
|
||||
if (event.type === "auto_retry_end") {
|
||||
retryEndEvents.push(event);
|
||||
}
|
||||
});
|
||||
|
||||
await expectPromptCompletes(session.prompt("think without delivering an answer"));
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(mock.calls).toHaveLength(4);
|
||||
expect(retryEndEvents).toHaveLength(1);
|
||||
expect(retryEndEvents[0]?.success).toBe(false);
|
||||
const finalError = retryEndEvents[0]?.finalError ?? "";
|
||||
expect(finalError).toContain("/shake images");
|
||||
expect(finalError).not.toContain("billed");
|
||||
});
|
||||
|
||||
it("keeps the context hint for a capped thinking-only stop even though it billed output", async () => {
|
||||
const { session, mock } = await createHarness([
|
||||
thinkingOnlyStop(),
|
||||
|
||||
Reference in New Issue
Block a user