fix(ai): retried empty ollama completions

- Wrapped the Ollama chat provider with the shared empty-completion retry layer.

- Treated EOS-only one-token empty stops as retryable while preserving visible-content commits.

- Added retry utility and Ollama provider regressions for output: 1 empty stops.

Fixes #4659
This commit is contained in:
roboomp
2026-07-06 02:31:23 +00:00
parent 79a397e97a
commit 25328dcedf
5 changed files with 72 additions and 7 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Ollama/Ollama Cloud EOS-only completions to retry empty stops with a single output token before the agent loop can halt silently. ([#4659](https://github.com/can1357/oh-my-pi/issues/4659))
## [16.3.7] - 2026-07-05
### Fixed
+7 -2
View File
@@ -16,6 +16,7 @@ import type {
} from "../types";
import { normalizeSystemPrompts } from "../utils";
import { clearStreamingPartialJson, kStreamingPartialJson } from "../utils/block-symbols";
import { withEmptyCompletionRetry } from "../utils/empty-completion-retry";
import { AssistantMessageEventStream } from "../utils/event-stream";
import type { CapturedHttpErrorResponse, RawHttpRequestDump } from "../utils/http-inspector";
import {
@@ -449,10 +450,10 @@ function hasVisibleAssistantContent(output: AssistantMessage): boolean {
const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000];
export const streamOllama: StreamFunction<"ollama-chat"> = (
const streamOllamaOnce = (
model: Model<"ollama-chat">,
context: Context,
options: OllamaChatOptions,
options: OllamaChatOptions = {},
): AssistantMessageEventStream => {
const stream = new AssistantMessageEventStream();
void (async () => {
@@ -771,3 +772,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
})();
return stream;
};
/** Retry EOS-only Ollama completions before the agent loop sees an empty stop. */
export const streamOllama: StreamFunction<"ollama-chat"> = (model, context, options) =>
withEmptyCompletionRetry(model, context, options, streamOllamaOnce);
@@ -109,17 +109,16 @@ export function withEmptyCompletionRetry<M, O extends EmptyCompletionRetryOption
}
// Retry only a genuinely degenerate completion: a normal stop that
// produced no visible content AND billed no output tokens (the flaky
// gateway signature — charged nothing, returned nothing). A stop that
// reports output tokens spent its budget somewhere (e.g. thinking) and
// is left alone.
// produced no visible content and reported no generated content tokens.
// Some providers count the terminal EOS as one output token, so a
// one-token invisible stop is still the same empty-completion failure.
const message = terminal?.type === "done" ? terminal.message : undefined;
const isRetryableEmpty =
!committed &&
message !== undefined &&
message.stopReason === "stop" &&
!message.errorMessage &&
(message.usage?.output ?? 0) <= 0 &&
(message.usage?.output ?? 0) <= 1 &&
!hasVisibleAssistantContent(message);
if (isRetryableEmpty && emptyAttempt < MAX_EMPTY_COMPLETION_RETRIES && !signal?.aborted) {
@@ -51,6 +51,17 @@ function emptyAttempt(): AssistantMessageEventStream {
] as unknown as AssistantMessageEvent[]);
}
/** start + stop with no visible content and a single EOS output token. */
function eosOnlyAttempt(): AssistantMessageEventStream {
const message = assistant();
message.usage.output = 1;
message.usage.totalTokens = 1;
return streamFromEvents([
{ type: "start", partial: message },
{ type: "done", reason: "stop", message },
] as unknown as AssistantMessageEvent[]);
}
function contentAttempt(): AssistantMessageEventStream {
const message = assistant(["hello"]);
return streamFromEvents([
@@ -89,6 +100,23 @@ describe("withEmptyCompletionRetry", () => {
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
});
it("retries an EOS-only empty stop that reports one output token", async () => {
let attempts = 0;
const waits: number[] = [];
const stream = withEmptyCompletionRetry({}, CTX, { providerRetryWait: async ms => void waits.push(ms) }, () => {
attempts++;
return attempts === 1 ? eosOnlyAttempt() : contentAttempt();
});
const events = await drain(stream);
const result = await stream.result();
expect(attempts).toBe(2);
expect(waits).toEqual([500]);
expect(events.filter(e => e.type === "start")).toHaveLength(1);
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
});
it("delivers the empty result after exhausting the retry cap", async () => {
let attempts = 0;
const waits: number[] = [];
@@ -81,6 +81,35 @@ describe("Ollama chat thinking controls", () => {
expect(payload?.think).toBe(false);
});
it("retries EOS-only empty completions before surfacing Ollama output", async () => {
let attempts = 0;
const fetchMock = async (): Promise<Response> => {
attempts++;
if (attempts === 1) {
return new Response(
'{"message":{"content":""},"done":true,"done_reason":"stop","prompt_eval_count":98563,"eval_count":1}\n',
{ status: 200 },
);
}
return new Response(
'{"message":{"content":"recovered"},"done":true,"done_reason":"stop","prompt_eval_count":98563,"eval_count":3}\n',
{ status: 200 },
);
};
const context: Context = {
messages: [{ role: "user", content: "Continue the task.", timestamp: 0 }],
};
const result = await streamOllama(createReasoningOllamaModel(), context, {
apiKey: "test-key",
fetch: fetchMock,
providerRetryWait: async () => {},
}).result();
expect(attempts).toBe(2);
expect(result.content).toEqual([{ type: "text", text: "recovered" }]);
});
it("normalizes tool schemas for Ollama's Go parser", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {