fix(ai): retried empty ollama completions
- Wrapped the Ollama chat provider with the shared empty-completion retry layer. - Treated EOS-only one-token empty stops as retryable while preserving visible-content commits. - Added retry utility and Ollama provider regressions for output: 1 empty stops. Fixes #4659
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama/Ollama Cloud EOS-only completions to retry empty stops with a single output token before the agent loop can halt silently. ([#4659](https://github.com/can1357/oh-my-pi/issues/4659))
|
||||
|
||||
## [16.3.7] - 2026-07-05
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -16,6 +16,7 @@ import type {
|
||||
} from "../types";
|
||||
import { normalizeSystemPrompts } from "../utils";
|
||||
import { clearStreamingPartialJson, kStreamingPartialJson } from "../utils/block-symbols";
|
||||
import { withEmptyCompletionRetry } from "../utils/empty-completion-retry";
|
||||
import { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import type { CapturedHttpErrorResponse, RawHttpRequestDump } from "../utils/http-inspector";
|
||||
import {
|
||||
@@ -449,10 +450,10 @@ function hasVisibleAssistantContent(output: AssistantMessage): boolean {
|
||||
|
||||
const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000];
|
||||
|
||||
export const streamOllama: StreamFunction<"ollama-chat"> = (
|
||||
const streamOllamaOnce = (
|
||||
model: Model<"ollama-chat">,
|
||||
context: Context,
|
||||
options: OllamaChatOptions,
|
||||
options: OllamaChatOptions = {},
|
||||
): AssistantMessageEventStream => {
|
||||
const stream = new AssistantMessageEventStream();
|
||||
void (async () => {
|
||||
@@ -771,3 +772,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
||||
})();
|
||||
return stream;
|
||||
};
|
||||
|
||||
/** Retry EOS-only Ollama completions before the agent loop sees an empty stop. */
|
||||
export const streamOllama: StreamFunction<"ollama-chat"> = (model, context, options) =>
|
||||
withEmptyCompletionRetry(model, context, options, streamOllamaOnce);
|
||||
|
||||
@@ -109,17 +109,16 @@ export function withEmptyCompletionRetry<M, O extends EmptyCompletionRetryOption
|
||||
}
|
||||
|
||||
// Retry only a genuinely degenerate completion: a normal stop that
|
||||
// produced no visible content AND billed no output tokens (the flaky
|
||||
// gateway signature — charged nothing, returned nothing). A stop that
|
||||
// reports output tokens spent its budget somewhere (e.g. thinking) and
|
||||
// is left alone.
|
||||
// produced no visible content and reported no generated content tokens.
|
||||
// Some providers count the terminal EOS as one output token, so a
|
||||
// one-token invisible stop is still the same empty-completion failure.
|
||||
const message = terminal?.type === "done" ? terminal.message : undefined;
|
||||
const isRetryableEmpty =
|
||||
!committed &&
|
||||
message !== undefined &&
|
||||
message.stopReason === "stop" &&
|
||||
!message.errorMessage &&
|
||||
(message.usage?.output ?? 0) <= 0 &&
|
||||
(message.usage?.output ?? 0) <= 1 &&
|
||||
!hasVisibleAssistantContent(message);
|
||||
|
||||
if (isRetryableEmpty && emptyAttempt < MAX_EMPTY_COMPLETION_RETRIES && !signal?.aborted) {
|
||||
|
||||
@@ -51,6 +51,17 @@ function emptyAttempt(): AssistantMessageEventStream {
|
||||
] as unknown as AssistantMessageEvent[]);
|
||||
}
|
||||
|
||||
/** start + stop with no visible content and a single EOS output token. */
|
||||
function eosOnlyAttempt(): AssistantMessageEventStream {
|
||||
const message = assistant();
|
||||
message.usage.output = 1;
|
||||
message.usage.totalTokens = 1;
|
||||
return streamFromEvents([
|
||||
{ type: "start", partial: message },
|
||||
{ type: "done", reason: "stop", message },
|
||||
] as unknown as AssistantMessageEvent[]);
|
||||
}
|
||||
|
||||
function contentAttempt(): AssistantMessageEventStream {
|
||||
const message = assistant(["hello"]);
|
||||
return streamFromEvents([
|
||||
@@ -89,6 +100,23 @@ describe("withEmptyCompletionRetry", () => {
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("retries an EOS-only empty stop that reports one output token", async () => {
|
||||
let attempts = 0;
|
||||
const waits: number[] = [];
|
||||
const stream = withEmptyCompletionRetry({}, CTX, { providerRetryWait: async ms => void waits.push(ms) }, () => {
|
||||
attempts++;
|
||||
return attempts === 1 ? eosOnlyAttempt() : contentAttempt();
|
||||
});
|
||||
|
||||
const events = await drain(stream);
|
||||
const result = await stream.result();
|
||||
|
||||
expect(attempts).toBe(2);
|
||||
expect(waits).toEqual([500]);
|
||||
expect(events.filter(e => e.type === "start")).toHaveLength(1);
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("delivers the empty result after exhausting the retry cap", async () => {
|
||||
let attempts = 0;
|
||||
const waits: number[] = [];
|
||||
|
||||
@@ -81,6 +81,35 @@ describe("Ollama chat thinking controls", () => {
|
||||
expect(payload?.think).toBe(false);
|
||||
});
|
||||
|
||||
it("retries EOS-only empty completions before surfacing Ollama output", async () => {
|
||||
let attempts = 0;
|
||||
const fetchMock = async (): Promise<Response> => {
|
||||
attempts++;
|
||||
if (attempts === 1) {
|
||||
return new Response(
|
||||
'{"message":{"content":""},"done":true,"done_reason":"stop","prompt_eval_count":98563,"eval_count":1}\n',
|
||||
{ status: 200 },
|
||||
);
|
||||
}
|
||||
return new Response(
|
||||
'{"message":{"content":"recovered"},"done":true,"done_reason":"stop","prompt_eval_count":98563,"eval_count":3}\n',
|
||||
{ status: 200 },
|
||||
);
|
||||
};
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "Continue the task.", timestamp: 0 }],
|
||||
};
|
||||
|
||||
const result = await streamOllama(createReasoningOllamaModel(), context, {
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
providerRetryWait: async () => {},
|
||||
}).result();
|
||||
|
||||
expect(attempts).toBe(2);
|
||||
expect(result.content).toEqual([{ type: "text", text: "recovered" }]);
|
||||
});
|
||||
|
||||
it("normalizes tool schemas for Ollama's Go parser", async () => {
|
||||
let payload: OllamaChatRequestPayload | undefined;
|
||||
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
|
||||
Reference in New Issue
Block a user