fix(ai): guarantee ollama user turn and surface done_reason load

An agent-attributed developer turn (e.g. a plan-approval handoff into a
fresh session) maps to Ollama's `system` role, so a request whose only
non-system message is that turn carried zero `user`-role turns. Ollama
answers such a request with `done_reason: "load"`, generating nothing,
and `mapDoneReason` laundered it into a clean `stopReason: "stop"` with
zero usage — indistinguishable from a legitimate empty completion, so
every recovery layer retried the impossible request until the cap
surfaced a misleading empty-stop error.

- convertMessages now demotes the last non-prefix `system` turn to
  `user` when no user turn survives, keeping the static system-prompt
  prefix intact for prefix caching.
- mapDoneReason maps `done_reason: "load"` to `error` with an explicit
  message so it surfaces immediately instead of being retried.

Fixes #7465
This commit is contained in:
roboomp
2026-08-03 07:33:22 +00:00
parent 5039b33a11
commit 2586c422ec
3 changed files with 119 additions and 1 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Ollama requests with no `user`-role turn (e.g. a plan-approval handoff whose only non-system message is an agent-attributed developer turn) returning `done_reason: "load"` and generating nothing, which was laundered into a clean empty stop and retried until the cap surfaced a misleading error. `convertMessages` now demotes the last non-prefix `system` turn to `user` so the request can produce output, and `mapDoneReason` surfaces `done_reason: "load"` as an error instead of a silent stop ([#7465](https://github.com/can1357/oh-my-pi/issues/7465)).
## [17.2.5] - 2026-08-03
### Changed
+27 -1
View File
@@ -251,7 +251,7 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM
const messages: Message[] = [...systemMessages, ...context.messages];
const isCloud = model.provider === "ollama-cloud";
const supportsImages = model.input.includes("image");
return transformMessages(messages, model).map((msg, index) => {
const converted = transformMessages(messages, model).map((msg, index) => {
// Real `systemPrompt` entries (always emitted first) stay on Ollama's
// `system` role. After the static prefix, a developer turn keeps `system`
// when it's an agent-owned control instruction (empty/unexpected-stop
@@ -272,6 +272,21 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM
}
return converted;
});
// Ollama returns `done_reason: "load"` and generates nothing when a request
// carries no `user`-role message (e.g. a plan-approval handoff into a fresh
// session whose only non-system turn is an agent-attributed developer message
// mapped to `system`). Demote the last non-prefix system turn to `user` so the
// request can actually produce output; the static system-prompt prefix stays
// on `system` for prefix caching. (#7465)
if (!converted.some(m => m.role === "user")) {
for (let i = converted.length - 1; i >= systemPrompts.length; i--) {
if (converted[i].role === "system") {
converted[i].role = "user";
break;
}
}
}
return converted;
}
function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefined {
@@ -403,6 +418,12 @@ function mapDoneReason(doneReason: string | undefined, output: AssistantMessage)
if (doneReason === "tool_calls") {
return "toolUse";
}
if (doneReason === "load") {
// Ollama emits done_reason:"load" (model loaded, nothing generated) when a
// request has no user-role turn. Surface it as an error rather than a clean
// empty stop so it isn't laundered and retried behind a misleading hint. (#7465)
return "error";
}
if (doneReason === undefined && output.content.some(block => block.type === "toolCall")) {
return "toolUse";
}
@@ -411,6 +432,8 @@ function mapDoneReason(doneReason: string | undefined, output: AssistantMessage)
const EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE =
"Model returned no content: prompt filled the context window; raise Ollama num_ctx or shorten the prompt.";
const EMPTY_OLLAMA_LOAD_COMPLETION_MESSAGE =
"Ollama loaded the model but generated nothing (done_reason: load): the request contained no user-role message.";
function hasVisibleAssistantContent(output: AssistantMessage): boolean {
return output.content.some(block => {
@@ -697,6 +720,9 @@ const streamOllamaOnce = (
output.stopReason = "error";
output.errorMessage = EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE;
}
if (output.stopReason === "error" && !output.errorMessage) {
output.errorMessage = EMPTY_OLLAMA_LOAD_COMPLETION_MESSAGE;
}
// Tool calls always mean "execute and continue" in the OpenAI/Ollama contract.
// If the turn produced tool-call blocks but reported a natural `stop`, promote
// to `toolUse` so the agent loop runs them (it gates execution on the stop
@@ -0,0 +1,88 @@
import { describe, expect, it } from "bun:test";
import type { Context } from "@oh-my-pi/pi-ai";
import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
interface OllamaChatMessagePayload {
role?: unknown;
content?: unknown;
}
interface OllamaChatRequestPayload {
messages?: OllamaChatMessagePayload[];
}
function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload {
if (value === null || typeof value !== "object") return false;
const payload = value as { messages?: unknown };
return payload.messages === undefined || Array.isArray(payload.messages);
}
function createOllamaModel() {
return buildModel({
id: "deepseek-v4-flash",
name: "DeepSeek V4 Flash",
api: "ollama-chat",
provider: "ollama-cloud",
baseUrl: "https://ollama.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 8192,
});
}
describe("Ollama no-user-turn handling", () => {
it("demotes the trailing agent developer turn to user when no user turn survives", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) throw new Error("Expected Ollama payload object");
payload = parsed;
return new Response(
'{"message":{"content":"ok"},"done":true,"done_reason":"stop","prompt_eval_count":9,"eval_count":2}\n',
{
status: 200,
},
);
};
const context: Context = {
systemPrompt: ["static system"],
messages: [
{
role: "developer",
content: [{ type: "text", text: "Plan approved. Read the plan and implement it now." }],
attribution: "agent",
timestamp: Date.now(),
},
],
};
await streamOllama(createOllamaModel(), context, { apiKey: "test-key", fetch: fetchMock }).result();
const roles = payload?.messages?.map(m => m.role);
expect(roles).toContain("user");
// The static system prefix stays on `system`.
expect(roles?.[0]).toBe("system");
expect(payload?.messages?.at(-1)?.role).toBe("user");
expect(payload?.messages?.at(-1)?.content).toContain("Plan approved.");
});
it('surfaces done_reason:"load" as an error instead of a clean empty stop', async () => {
const fetchMock = async (): Promise<Response> =>
new Response('{"message":{"content":""},"done":true,"done_reason":"load"}\n', { status: 200 });
const context: Context = {
messages: [{ role: "user", content: "hi", timestamp: 0 }],
};
const result = await streamOllama(createOllamaModel(), context, {
apiKey: "test-key",
fetch: fetchMock,
providerRetryWait: async () => {},
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage ?? "").toContain("load");
});
});