Merge remote-tracking branch 'origin/farm/3b99853a/fix-auto-learn-stop-cache'

This commit is contained in:
can1357
2026-06-25 12:44:37 +02:00
3 changed files with 115 additions and 17 deletions
+1
View File
@@ -4,6 +4,7 @@
### Fixed
- Fixed Ollama/llama.cpp chat payloads serializing user-attributed mid-conversation developer messages (auto-learn capture nudge, advisor cards, file-mention companions) as `system` turns; they now serialize as `user` so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. Agent-owned developer reminders (`attribution: "agent"` — empty/unexpected-stop retries, checkpoint rewind warning, todo reminders) keep their `system` priority. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456))
- Fixed prior-turn reasoning being lost on cross-API provider switches: when a session moved from an Anthropic-compatible 3p endpoint to an OpenAI-compatible one (Z.AI Anthropic → Z.AI OpenAI, Kimi Anthropic → Kimi OpenAI, DeepSeek, OpenCode-hosted reasoning models, or any custom `models.yaml` switch that crosses API types), the cross-API path of `transformMessages` text-demoted every prior `thinking` block, so the next request shipped the reasoning chain as plain conversation `content` instead of structured `reasoning_content` — losing it as reasoning context and re-billing it. `convertMessages` now threads the request-time resolved compat into `transformMessages`, which preserves the prior reasoning as a native, signature-stripped `thinking` block whenever that resolved target accepts `reasoning_content` as a continuation hint (`requiresReasoningContentForToolCalls` — including the `whenThinking` policy OpenCode reactivates for thinking-on requests, #1071/#1484 — or `thinkingFormat: "zai"`); the `openai-completions` encoder surfaces those blocks via `reasoningContentField`, with a new branch for Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi, Xiaomi MiMo) that accept but don't require the field. Targets that can't replay unsigned reasoning (encrypted reasoning blobs, signed thought parts, non-reasoning models, thinking-disabled OpenCode) still text-demote so the reasoning survives as conversation context. ([#3437](https://github.com/can1357/oh-my-pi/pull/3437), [#3439](https://github.com/can1357/oh-my-pi/pull/3439) by [@roboomp](https://github.com/roboomp); [#3433](https://github.com/can1357/oh-my-pi/issues/3433), [#3434](https://github.com/can1357/oh-my-pi/issues/3434))
- Fixed Bedrock cross-region inference profiles routing to `us-east-1` regardless of their geo prefix: a profile such as `eu.anthropic.claude-…` (or `apac.`/`au.`/`jp.`) sent to the hardcoded `us-east-1` endpoint returned HTTP 400 `The provided model identifier is invalid`. `streamBedrock` now derives the runtime region from the profile's geo prefix — honoring an ambient `AWS_REGION`/`AWS_DEFAULT_REGION` only when it can serve that geo and falling back to the geo's default region otherwise — while explicit per-request and ARN-embedded regions still win and region-agnostic `global.` profiles stay unchanged.
- Fixed malformed tool calls (empty `name`) wedging entire sessions in HTTP 400 loops: when a model occasionally emits `{ "name": "", "arguments": "{}" }` (observed: GLM-5.2 + thinking on long turns), the agent rejected the call at execution time with `Tool not found`, but the malformed block plus its error `toolResult` stayed in conversation history and every subsequent request 400'd on `tool_use.name`/`tool_calls[i].function.name` validation until the user ran `/clear`. `transformMessages` — the canonical sanitize boundary every provider passes through — now drops `toolCall` blocks with empty/whitespace `name`, pairs them with their `toolResult` messages only inside the same assistant→tool-result window (per-id FIFO queue cleared at non-result boundaries, so stale malformed calls without a result cannot consume later valid duplicate-id outputs), and drops the assistant turn when it has no replayable content left. Defensive (provider-agnostic, fires regardless of model), idempotent (no-op on a clean history), and self-healing (one round-trip after the fix lands sanitizes an already-poisoned session). ([#3458](https://github.com/can1357/oh-my-pi/issues/3458))
+25 -17
View File
@@ -192,14 +192,18 @@ function toPlainContent(
};
}
function convertMessage(message: Message, supportsImages: boolean): OllamaMessage {
function convertMessage(
message: Message,
supportsImages: boolean,
developerRole: "system" | "user" = "user",
): OllamaMessage {
if (message.role === "user") {
const converted = toPlainContent(message.content, supportsImages);
return { role: "user", ...converted };
}
if (message.role === "developer") {
const converted = toPlainContent(message.content, supportsImages);
return { role: "system", ...converted };
return { role: developerRole, ...converted };
}
if (message.role === "toolResult") {
const converted = toPlainContent(message.content, supportsImages);
@@ -240,23 +244,27 @@ function convertMessage(message: Message, supportsImages: boolean): OllamaMessag
}
function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaMessage[] {
const messages: Message[] = [];
// Emit one developer message per ordered system prompt. The wire role is mapped to "system"
// by `convertMessage`, but keeping the prompts separate preserves prefix-cache stability:
// if only the trailing prompt changes between calls, the leading system messages keep
// their identical token prefix so KV-cache reuse covers them.
for (const systemPrompt of normalizeSystemPrompts(context.systemPrompt)) {
messages.push({
role: "developer",
content: systemPrompt,
timestamp: Date.now(),
});
}
messages.push(...context.messages);
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
const systemMessages: Message[] = systemPrompts.map(systemPrompt => ({
role: "developer",
content: systemPrompt,
timestamp: Date.now(),
}));
const messages: Message[] = [...systemMessages, ...context.messages];
const isCloud = model.provider === "ollama-cloud";
const supportsImages = model.input.includes("image");
return transformMessages(messages, model).map(msg => {
const converted = convertMessage(msg, supportsImages);
return transformMessages(messages, model).map((msg, index) => {
// Real `systemPrompt` entries (always emitted first) stay on Ollama's
// `system` role. After the static prefix, a developer turn keeps `system`
// when it's an agent-owned control instruction (empty/unexpected-stop
// retries, checkpoint rewind warning, todo reminders — all carry
// `attribution: "agent"`), but a user-attributed developer turn (auto-learn
// capture nudge, advisor cards, file-mention companions) drops to `user`.
// That keeps the in-conversation byte prefix stable for prefix caches
// (llama.cpp, #3456) without demoting mandatory agent reminders.
const developerRole =
msg.role === "developer" && (index < systemPrompts.length || msg.attribution !== "user") ? "system" : "user";
const converted = convertMessage(msg, supportsImages, developerRole);
// Ollama cloud rejects requests when assistant history messages contain the `thinking`
// field — it's valid in model responses but not accepted as a history input. Strip it
// to prevent HTTP 400 errors. Local Ollama instances are unaffected.
@@ -70,6 +70,95 @@ describe("Ollama chat thinking controls", () => {
expect(payload?.think).toBe(false);
});
it("sends mid-conversation developer messages as user turns for llama.cpp cache reuse", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"captured"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const now = Date.now();
const context: Context = {
systemPrompt: ["static system"],
messages: [
{ role: "user", content: "Do work", timestamp: now - 2 },
{
role: "assistant",
content: [{ type: "text", text: "Done" }],
api: "ollama-chat",
provider: "ollama",
model: "llama",
usage: emptyUsage,
stopReason: "stop",
timestamp: now - 1,
},
{
role: "developer",
content: [{ type: "text", text: "Capture reusable lessons." }],
attribution: "user",
timestamp: now,
},
],
};
await streamOllama(createReasoningOllamaModel(), context, {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "user"]);
expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons.");
});
it("keeps agent-attributed developer reminders on the system role", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"resumed"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const now = Date.now();
const context: Context = {
systemPrompt: ["static system"],
messages: [
{ role: "user", content: "Do work", timestamp: now - 2 },
{
role: "assistant",
content: [{ type: "text", text: "Done" }],
api: "ollama-chat",
provider: "ollama",
model: "llama",
usage: emptyUsage,
stopReason: "stop",
timestamp: now - 1,
},
{
role: "developer",
content: [{ type: "text", text: "<system-warning>complete the checkpoint</system-warning>" }],
attribution: "agent",
timestamp: now,
},
],
};
await streamOllama(createReasoningOllamaModel(), context, {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "system"]);
expect(payload?.messages?.at(-1)?.content).toContain("complete the checkpoint");
});
it("omits tool-result images for text-only Ollama chat models", async () => {
let payload: OllamaChatRequestPayload | undefined;