diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f37040276..a740add37 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Ollama requests with no `user`-role turn (e.g. a plan-approval handoff whose only non-system message is an agent-attributed developer turn) returning `done_reason: "load"` and generating nothing, which was laundered into a clean empty stop and retried until the cap surfaced a misleading error. `convertMessages` now demotes the last non-prefix `system` turn to `user` so the request can produce output, and `mapDoneReason` surfaces `done_reason: "load"` as an error instead of a silent stop ([#7465](https://github.com/can1357/oh-my-pi/issues/7465)). + ## [17.2.5] - 2026-08-03 ### Changed diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index aa3f1c672..f6e0b1e7d 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -251,7 +251,7 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM const messages: Message[] = [...systemMessages, ...context.messages]; const isCloud = model.provider === "ollama-cloud"; const supportsImages = model.input.includes("image"); - return transformMessages(messages, model).map((msg, index) => { + const converted = transformMessages(messages, model).map((msg, index) => { // Real `systemPrompt` entries (always emitted first) stay on Ollama's // `system` role. After the static prefix, a developer turn keeps `system` // when it's an agent-owned control instruction (empty/unexpected-stop @@ -272,6 +272,21 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM } return converted; }); + // Ollama returns `done_reason: "load"` and generates nothing when a request + // carries no `user`-role message (e.g. a plan-approval handoff into a fresh + // session whose only non-system turn is an agent-attributed developer message + // mapped to `system`). Demote the last non-prefix system turn to `user` so the + // request can actually produce output; the static system-prompt prefix stays + // on `system` for prefix caching. (#7465) + if (!converted.some(m => m.role === "user")) { + for (let i = converted.length - 1; i >= systemPrompts.length; i--) { + if (converted[i].role === "system") { + converted[i].role = "user"; + break; + } + } + } + return converted; } function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefined { @@ -403,6 +418,12 @@ function mapDoneReason(doneReason: string | undefined, output: AssistantMessage) if (doneReason === "tool_calls") { return "toolUse"; } + if (doneReason === "load") { + // Ollama emits done_reason:"load" (model loaded, nothing generated) when a + // request has no user-role turn. Surface it as an error rather than a clean + // empty stop so it isn't laundered and retried behind a misleading hint. (#7465) + return "error"; + } if (doneReason === undefined && output.content.some(block => block.type === "toolCall")) { return "toolUse"; } @@ -411,6 +432,8 @@ function mapDoneReason(doneReason: string | undefined, output: AssistantMessage) const EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE = "Model returned no content: prompt filled the context window; raise Ollama num_ctx or shorten the prompt."; +const EMPTY_OLLAMA_LOAD_COMPLETION_MESSAGE = + "Ollama loaded the model but generated nothing (done_reason: load): the request contained no user-role message."; function hasVisibleAssistantContent(output: AssistantMessage): boolean { return output.content.some(block => { @@ -697,6 +720,9 @@ const streamOllamaOnce = ( output.stopReason = "error"; output.errorMessage = EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE; } + if (output.stopReason === "error" && !output.errorMessage) { + output.errorMessage = EMPTY_OLLAMA_LOAD_COMPLETION_MESSAGE; + } // Tool calls always mean "execute and continue" in the OpenAI/Ollama contract. // If the turn produced tool-call blocks but reported a natural `stop`, promote // to `toolUse` so the agent loop runs them (it gates execution on the stop diff --git a/packages/ai/test/ollama-no-user-turn.test.ts b/packages/ai/test/ollama-no-user-turn.test.ts new file mode 100644 index 000000000..7abfe859e --- /dev/null +++ b/packages/ai/test/ollama-no-user-turn.test.ts @@ -0,0 +1,88 @@ +import { describe, expect, it } from "bun:test"; +import type { Context } from "@oh-my-pi/pi-ai"; +import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +interface OllamaChatMessagePayload { + role?: unknown; + content?: unknown; +} + +interface OllamaChatRequestPayload { + messages?: OllamaChatMessagePayload[]; +} + +function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload { + if (value === null || typeof value !== "object") return false; + const payload = value as { messages?: unknown }; + return payload.messages === undefined || Array.isArray(payload.messages); +} + +function createOllamaModel() { + return buildModel({ + id: "deepseek-v4-flash", + name: "DeepSeek V4 Flash", + api: "ollama-chat", + provider: "ollama-cloud", + baseUrl: "https://ollama.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8192, + }); +} + +describe("Ollama no-user-turn handling", () => { + it("demotes the trailing agent developer turn to user when no user turn survives", async () => { + let payload: OllamaChatRequestPayload | undefined; + const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { + const parsed: unknown = JSON.parse(String(init?.body)); + if (!isOllamaChatRequestPayload(parsed)) throw new Error("Expected Ollama payload object"); + payload = parsed; + return new Response( + '{"message":{"content":"ok"},"done":true,"done_reason":"stop","prompt_eval_count":9,"eval_count":2}\n', + { + status: 200, + }, + ); + }; + const context: Context = { + systemPrompt: ["static system"], + messages: [ + { + role: "developer", + content: [{ type: "text", text: "Plan approved. Read the plan and implement it now." }], + attribution: "agent", + timestamp: Date.now(), + }, + ], + }; + + await streamOllama(createOllamaModel(), context, { apiKey: "test-key", fetch: fetchMock }).result(); + + const roles = payload?.messages?.map(m => m.role); + expect(roles).toContain("user"); + // The static system prefix stays on `system`. + expect(roles?.[0]).toBe("system"); + expect(payload?.messages?.at(-1)?.role).toBe("user"); + expect(payload?.messages?.at(-1)?.content).toContain("Plan approved."); + }); + + it('surfaces done_reason:"load" as an error instead of a clean empty stop', async () => { + const fetchMock = async (): Promise => + new Response('{"message":{"content":""},"done":true,"done_reason":"load"}\n', { status: 200 }); + const context: Context = { + messages: [{ role: "user", content: "hi", timestamp: 0 }], + }; + + const result = await streamOllama(createOllamaModel(), context, { + apiKey: "test-key", + fetch: fetchMock, + providerRetryWait: async () => {}, + }).result(); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage ?? "").toContain("load"); + }); +});