diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2319ed0d0..3d1ceda1b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed Ollama/llama.cpp chat payloads serializing user-attributed mid-conversation developer messages (auto-learn capture nudge, advisor cards, file-mention companions) as `system` turns; they now serialize as `user` so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. Agent-owned developer reminders (`attribution: "agent"` — empty/unexpected-stop retries, checkpoint rewind warning, todo reminders) keep their `system` priority. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456)) - Fixed prior-turn reasoning being lost on cross-API provider switches: when a session moved from an Anthropic-compatible 3p endpoint to an OpenAI-compatible one (Z.AI Anthropic → Z.AI OpenAI, Kimi Anthropic → Kimi OpenAI, DeepSeek, OpenCode-hosted reasoning models, or any custom `models.yaml` switch that crosses API types), the cross-API path of `transformMessages` text-demoted every prior `thinking` block, so the next request shipped the reasoning chain as plain conversation `content` instead of structured `reasoning_content` — losing it as reasoning context and re-billing it. `convertMessages` now threads the request-time resolved compat into `transformMessages`, which preserves the prior reasoning as a native, signature-stripped `thinking` block whenever that resolved target accepts `reasoning_content` as a continuation hint (`requiresReasoningContentForToolCalls` — including the `whenThinking` policy OpenCode reactivates for thinking-on requests, #1071/#1484 — or `thinkingFormat: "zai"`); the `openai-completions` encoder surfaces those blocks via `reasoningContentField`, with a new branch for Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi, Xiaomi MiMo) that accept but don't require the field. Targets that can't replay unsigned reasoning (encrypted reasoning blobs, signed thought parts, non-reasoning models, thinking-disabled OpenCode) still text-demote so the reasoning survives as conversation context. ([#3437](https://github.com/can1357/oh-my-pi/pull/3437), [#3439](https://github.com/can1357/oh-my-pi/pull/3439) by [@roboomp](https://github.com/roboomp); [#3433](https://github.com/can1357/oh-my-pi/issues/3433), [#3434](https://github.com/can1357/oh-my-pi/issues/3434)) - Fixed Bedrock cross-region inference profiles routing to `us-east-1` regardless of their geo prefix: a profile such as `eu.anthropic.claude-…` (or `apac.`/`au.`/`jp.`) sent to the hardcoded `us-east-1` endpoint returned HTTP 400 `The provided model identifier is invalid`. `streamBedrock` now derives the runtime region from the profile's geo prefix — honoring an ambient `AWS_REGION`/`AWS_DEFAULT_REGION` only when it can serve that geo and falling back to the geo's default region otherwise — while explicit per-request and ARN-embedded regions still win and region-agnostic `global.` profiles stay unchanged. - Fixed malformed tool calls (empty `name`) wedging entire sessions in HTTP 400 loops: when a model occasionally emits `{ "name": "", "arguments": "{}" }` (observed: GLM-5.2 + thinking on long turns), the agent rejected the call at execution time with `Tool not found`, but the malformed block plus its error `toolResult` stayed in conversation history and every subsequent request 400'd on `tool_use.name`/`tool_calls[i].function.name` validation until the user ran `/clear`. `transformMessages` — the canonical sanitize boundary every provider passes through — now drops `toolCall` blocks with empty/whitespace `name`, pairs them with their `toolResult` messages only inside the same assistant→tool-result window (per-id FIFO queue cleared at non-result boundaries, so stale malformed calls without a result cannot consume later valid duplicate-id outputs), and drops the assistant turn when it has no replayable content left. Defensive (provider-agnostic, fires regardless of model), idempotent (no-op on a clean history), and self-healing (one round-trip after the fix lands sanitizes an already-poisoned session). ([#3458](https://github.com/can1357/oh-my-pi/issues/3458)) diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index 9701d83c9..0080eed46 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -192,14 +192,18 @@ function toPlainContent( }; } -function convertMessage(message: Message, supportsImages: boolean): OllamaMessage { +function convertMessage( + message: Message, + supportsImages: boolean, + developerRole: "system" | "user" = "user", +): OllamaMessage { if (message.role === "user") { const converted = toPlainContent(message.content, supportsImages); return { role: "user", ...converted }; } if (message.role === "developer") { const converted = toPlainContent(message.content, supportsImages); - return { role: "system", ...converted }; + return { role: developerRole, ...converted }; } if (message.role === "toolResult") { const converted = toPlainContent(message.content, supportsImages); @@ -240,23 +244,27 @@ function convertMessage(message: Message, supportsImages: boolean): OllamaMessag } function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaMessage[] { - const messages: Message[] = []; - // Emit one developer message per ordered system prompt. The wire role is mapped to "system" - // by `convertMessage`, but keeping the prompts separate preserves prefix-cache stability: - // if only the trailing prompt changes between calls, the leading system messages keep - // their identical token prefix so KV-cache reuse covers them. - for (const systemPrompt of normalizeSystemPrompts(context.systemPrompt)) { - messages.push({ - role: "developer", - content: systemPrompt, - timestamp: Date.now(), - }); - } - messages.push(...context.messages); + const systemPrompts = normalizeSystemPrompts(context.systemPrompt); + const systemMessages: Message[] = systemPrompts.map(systemPrompt => ({ + role: "developer", + content: systemPrompt, + timestamp: Date.now(), + })); + const messages: Message[] = [...systemMessages, ...context.messages]; const isCloud = model.provider === "ollama-cloud"; const supportsImages = model.input.includes("image"); - return transformMessages(messages, model).map(msg => { - const converted = convertMessage(msg, supportsImages); + return transformMessages(messages, model).map((msg, index) => { + // Real `systemPrompt` entries (always emitted first) stay on Ollama's + // `system` role. After the static prefix, a developer turn keeps `system` + // when it's an agent-owned control instruction (empty/unexpected-stop + // retries, checkpoint rewind warning, todo reminders — all carry + // `attribution: "agent"`), but a user-attributed developer turn (auto-learn + // capture nudge, advisor cards, file-mention companions) drops to `user`. + // That keeps the in-conversation byte prefix stable for prefix caches + // (llama.cpp, #3456) without demoting mandatory agent reminders. + const developerRole = + msg.role === "developer" && (index < systemPrompts.length || msg.attribution !== "user") ? "system" : "user"; + const converted = convertMessage(msg, supportsImages, developerRole); // Ollama cloud rejects requests when assistant history messages contain the `thinking` // field — it's valid in model responses but not accepted as a history input. Strip it // to prevent HTTP 400 errors. Local Ollama instances are unaffected. diff --git a/packages/ai/test/ollama-thinking-disable.test.ts b/packages/ai/test/ollama-thinking-disable.test.ts index 64ba28da9..bbdd92dd6 100644 --- a/packages/ai/test/ollama-thinking-disable.test.ts +++ b/packages/ai/test/ollama-thinking-disable.test.ts @@ -70,6 +70,95 @@ describe("Ollama chat thinking controls", () => { expect(payload?.think).toBe(false); }); + it("sends mid-conversation developer messages as user turns for llama.cpp cache reuse", async () => { + let payload: OllamaChatRequestPayload | undefined; + const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { + const parsed: unknown = JSON.parse(String(init?.body)); + if (!isOllamaChatRequestPayload(parsed)) { + throw new Error("Expected Ollama payload object"); + } + payload = parsed; + return new Response('{"message":{"content":"captured"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', { + status: 200, + }); + }; + const now = Date.now(); + const context: Context = { + systemPrompt: ["static system"], + messages: [ + { role: "user", content: "Do work", timestamp: now - 2 }, + { + role: "assistant", + content: [{ type: "text", text: "Done" }], + api: "ollama-chat", + provider: "ollama", + model: "llama", + usage: emptyUsage, + stopReason: "stop", + timestamp: now - 1, + }, + { + role: "developer", + content: [{ type: "text", text: "Capture reusable lessons." }], + attribution: "user", + timestamp: now, + }, + ], + }; + + await streamOllama(createReasoningOllamaModel(), context, { + apiKey: "test-key", + fetch: fetchMock, + }).result(); + + expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "user"]); + expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons."); + }); + + it("keeps agent-attributed developer reminders on the system role", async () => { + let payload: OllamaChatRequestPayload | undefined; + const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { + const parsed: unknown = JSON.parse(String(init?.body)); + if (!isOllamaChatRequestPayload(parsed)) { + throw new Error("Expected Ollama payload object"); + } + payload = parsed; + return new Response('{"message":{"content":"resumed"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', { + status: 200, + }); + }; + const now = Date.now(); + const context: Context = { + systemPrompt: ["static system"], + messages: [ + { role: "user", content: "Do work", timestamp: now - 2 }, + { + role: "assistant", + content: [{ type: "text", text: "Done" }], + api: "ollama-chat", + provider: "ollama", + model: "llama", + usage: emptyUsage, + stopReason: "stop", + timestamp: now - 1, + }, + { + role: "developer", + content: [{ type: "text", text: "complete the checkpoint" }], + attribution: "agent", + timestamp: now, + }, + ], + }; + + await streamOllama(createReasoningOllamaModel(), context, { + apiKey: "test-key", + fetch: fetchMock, + }).result(); + + expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "system"]); + expect(payload?.messages?.at(-1)?.content).toContain("complete the checkpoint"); + }); it("omits tool-result images for text-only Ollama chat models", async () => { let payload: OllamaChatRequestPayload | undefined;