From 3db3d6d1353304026dd30f8a7360b9b15bc70703 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 25 Jun 2026 08:46:53 +0000 Subject: [PATCH 1/3] fix(ai): preserved ollama cache for capture turns Serialize Ollama context developer messages after the static prompt as user turns so llama.cpp can reuse its warm prefix for auto-learn capture follow-ups. Fixes #3456 --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/providers/ollama.ts | 29 +++++------- .../ai/test/ollama-thinking-disable.test.ts | 45 +++++++++++++++++++ 3 files changed, 58 insertions(+), 17 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9adf5c474..a4dd2e8ed 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed Ollama/llama.cpp chat payloads serializing mid-conversation developer messages as `system` turns; follow-up capture nudges now remain normal `user` turns so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456)) - Fixed prior-turn reasoning being lost on cross-API provider switches: when a session moved from an Anthropic-compatible 3p endpoint to an OpenAI-compatible one (Z.AI Anthropic → Z.AI OpenAI, Kimi Anthropic → Kimi OpenAI, DeepSeek, OpenCode-hosted reasoning models, or any custom `models.yaml` switch that crosses API types), the cross-API path of `transformMessages` text-demoted every prior `thinking` block, so the next request shipped the reasoning chain as plain conversation `content` instead of structured `reasoning_content` — losing it as reasoning context and re-billing it. `convertMessages` now threads the request-time resolved compat into `transformMessages`, which preserves the prior reasoning as a native, signature-stripped `thinking` block whenever that resolved target accepts `reasoning_content` as a continuation hint (`requiresReasoningContentForToolCalls` — including the `whenThinking` policy OpenCode reactivates for thinking-on requests, #1071/#1484 — or `thinkingFormat: "zai"`); the `openai-completions` encoder surfaces those blocks via `reasoningContentField`, with a new branch for Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi, Xiaomi MiMo) that accept but don't require the field. Targets that can't replay unsigned reasoning (encrypted reasoning blobs, signed thought parts, non-reasoning models, thinking-disabled OpenCode) still text-demote so the reasoning survives as conversation context. ([#3437](https://github.com/can1357/oh-my-pi/pull/3437), [#3439](https://github.com/can1357/oh-my-pi/pull/3439) by [@roboomp](https://github.com/roboomp); [#3433](https://github.com/can1357/oh-my-pi/issues/3433), [#3434](https://github.com/can1357/oh-my-pi/issues/3434)) ## [16.1.18] - 2026-06-25 diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index 9701d83c9..e74fd015b 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -192,14 +192,14 @@ function toPlainContent( }; } -function convertMessage(message: Message, supportsImages: boolean): OllamaMessage { +function convertMessage(message: Message, supportsImages: boolean, developerRole: "system" | "user" = "user"): OllamaMessage { if (message.role === "user") { const converted = toPlainContent(message.content, supportsImages); return { role: "user", ...converted }; } if (message.role === "developer") { const converted = toPlainContent(message.content, supportsImages); - return { role: "system", ...converted }; + return { role: developerRole, ...converted }; } if (message.role === "toolResult") { const converted = toPlainContent(message.content, supportsImages); @@ -240,23 +240,18 @@ function convertMessage(message: Message, supportsImages: boolean): OllamaMessag } function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaMessage[] { - const messages: Message[] = []; - // Emit one developer message per ordered system prompt. The wire role is mapped to "system" - // by `convertMessage`, but keeping the prompts separate preserves prefix-cache stability: - // if only the trailing prompt changes between calls, the leading system messages keep - // their identical token prefix so KV-cache reuse covers them. - for (const systemPrompt of normalizeSystemPrompts(context.systemPrompt)) { - messages.push({ - role: "developer", - content: systemPrompt, - timestamp: Date.now(), - }); - } - messages.push(...context.messages); + const systemPrompts = normalizeSystemPrompts(context.systemPrompt); + const systemMessages: Message[] = systemPrompts.map(systemPrompt => ({ + role: "developer", + content: systemPrompt, + timestamp: Date.now(), + })); + const messages: Message[] = [...systemMessages, ...context.messages]; const isCloud = model.provider === "ollama-cloud"; const supportsImages = model.input.includes("image"); - return transformMessages(messages, model).map(msg => { - const converted = convertMessage(msg, supportsImages); + return transformMessages(messages, model).map((msg, index) => { + const developerRole = msg.role === "developer" && index < systemPrompts.length ? "system" : "user"; + const converted = convertMessage(msg, supportsImages, developerRole); // Ollama cloud rejects requests when assistant history messages contain the `thinking` // field — it's valid in model responses but not accepted as a history input. Strip it // to prevent HTTP 400 errors. Local Ollama instances are unaffected. diff --git a/packages/ai/test/ollama-thinking-disable.test.ts b/packages/ai/test/ollama-thinking-disable.test.ts index 64ba28da9..d87ff812e 100644 --- a/packages/ai/test/ollama-thinking-disable.test.ts +++ b/packages/ai/test/ollama-thinking-disable.test.ts @@ -70,6 +70,51 @@ describe("Ollama chat thinking controls", () => { expect(payload?.think).toBe(false); }); + it("sends mid-conversation developer messages as user turns for llama.cpp cache reuse", async () => { + let payload: OllamaChatRequestPayload | undefined; + const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { + const parsed: unknown = JSON.parse(String(init?.body)); + if (!isOllamaChatRequestPayload(parsed)) { + throw new Error("Expected Ollama payload object"); + } + payload = parsed; + return new Response('{"message":{"content":"captured"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', { + status: 200, + }); + }; + const now = Date.now(); + const context: Context = { + systemPrompt: ["static system"], + messages: [ + { role: "user", content: "Do work", timestamp: now - 2 }, + { + role: "assistant", + content: [{ type: "text", text: "Done" }], + api: "ollama-chat", + provider: "ollama", + model: "llama", + usage: emptyUsage, + stopReason: "stop", + timestamp: now - 1, + }, + { + role: "developer", + content: [{ type: "text", text: "Capture reusable lessons." }], + attribution: "user", + timestamp: now, + }, + ], + }; + + await streamOllama(createReasoningOllamaModel(), context, { + apiKey: "test-key", + fetch: fetchMock, + }).result(); + + expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "user"]); + expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons."); + }); + it("omits tool-result images for text-only Ollama chat models", async () => { let payload: OllamaChatRequestPayload | undefined; From 208d24de6aec61a12756e7ed6cbaac6e8a21cafb Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 25 Jun 2026 08:47:06 +0000 Subject: [PATCH 2/3] style: bun run fix --- packages/ai/src/providers/ollama.ts | 6 +++++- packages/ai/test/ollama-thinking-disable.test.ts | 1 - 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index e74fd015b..53a0ed348 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -192,7 +192,11 @@ function toPlainContent( }; } -function convertMessage(message: Message, supportsImages: boolean, developerRole: "system" | "user" = "user"): OllamaMessage { +function convertMessage( + message: Message, + supportsImages: boolean, + developerRole: "system" | "user" = "user", +): OllamaMessage { if (message.role === "user") { const converted = toPlainContent(message.content, supportsImages); return { role: "user", ...converted }; diff --git a/packages/ai/test/ollama-thinking-disable.test.ts b/packages/ai/test/ollama-thinking-disable.test.ts index d87ff812e..00104c888 100644 --- a/packages/ai/test/ollama-thinking-disable.test.ts +++ b/packages/ai/test/ollama-thinking-disable.test.ts @@ -115,7 +115,6 @@ describe("Ollama chat thinking controls", () => { expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons."); }); - it("omits tool-result images for text-only Ollama chat models", async () => { let payload: OllamaChatRequestPayload | undefined; const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { From f9cb0093246f60684613a3e6d1e96b1f564fc3bc Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 25 Jun 2026 08:54:18 +0000 Subject: [PATCH 3/3] fix(ai): kept agent-owned developer reminders on system Route mid-conversation developer messages by attribution: only user-attributed entries (auto-learn nudge, advisor cards, file mentions) drop to user; agent-owned reminders (retry, checkpoint, todo) stay on system role. Refs #3456 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/ollama.ts | 11 ++++- .../ai/test/ollama-thinking-disable.test.ts | 45 +++++++++++++++++++ 3 files changed, 56 insertions(+), 2 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index a4dd2e8ed..329b34d95 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Ollama/llama.cpp chat payloads serializing mid-conversation developer messages as `system` turns; follow-up capture nudges now remain normal `user` turns so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456)) +- Fixed Ollama/llama.cpp chat payloads serializing user-attributed mid-conversation developer messages (auto-learn capture nudge, advisor cards, file-mention companions) as `system` turns; they now serialize as `user` so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. Agent-owned developer reminders (`attribution: "agent"` — empty/unexpected-stop retries, checkpoint rewind warning, todo reminders) keep their `system` priority. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456)) - Fixed prior-turn reasoning being lost on cross-API provider switches: when a session moved from an Anthropic-compatible 3p endpoint to an OpenAI-compatible one (Z.AI Anthropic → Z.AI OpenAI, Kimi Anthropic → Kimi OpenAI, DeepSeek, OpenCode-hosted reasoning models, or any custom `models.yaml` switch that crosses API types), the cross-API path of `transformMessages` text-demoted every prior `thinking` block, so the next request shipped the reasoning chain as plain conversation `content` instead of structured `reasoning_content` — losing it as reasoning context and re-billing it. `convertMessages` now threads the request-time resolved compat into `transformMessages`, which preserves the prior reasoning as a native, signature-stripped `thinking` block whenever that resolved target accepts `reasoning_content` as a continuation hint (`requiresReasoningContentForToolCalls` — including the `whenThinking` policy OpenCode reactivates for thinking-on requests, #1071/#1484 — or `thinkingFormat: "zai"`); the `openai-completions` encoder surfaces those blocks via `reasoningContentField`, with a new branch for Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi, Xiaomi MiMo) that accept but don't require the field. Targets that can't replay unsigned reasoning (encrypted reasoning blobs, signed thought parts, non-reasoning models, thinking-disabled OpenCode) still text-demote so the reasoning survives as conversation context. ([#3437](https://github.com/can1357/oh-my-pi/pull/3437), [#3439](https://github.com/can1357/oh-my-pi/pull/3439) by [@roboomp](https://github.com/roboomp); [#3433](https://github.com/can1357/oh-my-pi/issues/3433), [#3434](https://github.com/can1357/oh-my-pi/issues/3434)) ## [16.1.18] - 2026-06-25 diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index 53a0ed348..0080eed46 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -254,7 +254,16 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM const isCloud = model.provider === "ollama-cloud"; const supportsImages = model.input.includes("image"); return transformMessages(messages, model).map((msg, index) => { - const developerRole = msg.role === "developer" && index < systemPrompts.length ? "system" : "user"; + // Real `systemPrompt` entries (always emitted first) stay on Ollama's + // `system` role. After the static prefix, a developer turn keeps `system` + // when it's an agent-owned control instruction (empty/unexpected-stop + // retries, checkpoint rewind warning, todo reminders — all carry + // `attribution: "agent"`), but a user-attributed developer turn (auto-learn + // capture nudge, advisor cards, file-mention companions) drops to `user`. + // That keeps the in-conversation byte prefix stable for prefix caches + // (llama.cpp, #3456) without demoting mandatory agent reminders. + const developerRole = + msg.role === "developer" && (index < systemPrompts.length || msg.attribution !== "user") ? "system" : "user"; const converted = convertMessage(msg, supportsImages, developerRole); // Ollama cloud rejects requests when assistant history messages contain the `thinking` // field — it's valid in model responses but not accepted as a history input. Strip it diff --git a/packages/ai/test/ollama-thinking-disable.test.ts b/packages/ai/test/ollama-thinking-disable.test.ts index 00104c888..bbdd92dd6 100644 --- a/packages/ai/test/ollama-thinking-disable.test.ts +++ b/packages/ai/test/ollama-thinking-disable.test.ts @@ -115,6 +115,51 @@ describe("Ollama chat thinking controls", () => { expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons."); }); + it("keeps agent-attributed developer reminders on the system role", async () => { + let payload: OllamaChatRequestPayload | undefined; + const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => { + const parsed: unknown = JSON.parse(String(init?.body)); + if (!isOllamaChatRequestPayload(parsed)) { + throw new Error("Expected Ollama payload object"); + } + payload = parsed; + return new Response('{"message":{"content":"resumed"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', { + status: 200, + }); + }; + const now = Date.now(); + const context: Context = { + systemPrompt: ["static system"], + messages: [ + { role: "user", content: "Do work", timestamp: now - 2 }, + { + role: "assistant", + content: [{ type: "text", text: "Done" }], + api: "ollama-chat", + provider: "ollama", + model: "llama", + usage: emptyUsage, + stopReason: "stop", + timestamp: now - 1, + }, + { + role: "developer", + content: [{ type: "text", text: "complete the checkpoint" }], + attribution: "agent", + timestamp: now, + }, + ], + }; + + await streamOllama(createReasoningOllamaModel(), context, { + apiKey: "test-key", + fetch: fetchMock, + }).result(); + + expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "system"]); + expect(payload?.messages?.at(-1)?.content).toContain("complete the checkpoint"); + }); + it("omits tool-result images for text-only Ollama chat models", async () => { let payload: OllamaChatRequestPayload | undefined; const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise => {