Merge remote-tracking branch 'origin/farm/3b99853a/fix-auto-learn-stop-cache'
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama/llama.cpp chat payloads serializing user-attributed mid-conversation developer messages (auto-learn capture nudge, advisor cards, file-mention companions) as `system` turns; they now serialize as `user` so llama.cpp can reuse the warm prompt prefix instead of forcing full re-processing. Agent-owned developer reminders (`attribution: "agent"` — empty/unexpected-stop retries, checkpoint rewind warning, todo reminders) keep their `system` priority. ([#3456](https://github.com/can1357/oh-my-pi/issues/3456))
|
||||
- Fixed prior-turn reasoning being lost on cross-API provider switches: when a session moved from an Anthropic-compatible 3p endpoint to an OpenAI-compatible one (Z.AI Anthropic → Z.AI OpenAI, Kimi Anthropic → Kimi OpenAI, DeepSeek, OpenCode-hosted reasoning models, or any custom `models.yaml` switch that crosses API types), the cross-API path of `transformMessages` text-demoted every prior `thinking` block, so the next request shipped the reasoning chain as plain conversation `content` instead of structured `reasoning_content` — losing it as reasoning context and re-billing it. `convertMessages` now threads the request-time resolved compat into `transformMessages`, which preserves the prior reasoning as a native, signature-stripped `thinking` block whenever that resolved target accepts `reasoning_content` as a continuation hint (`requiresReasoningContentForToolCalls` — including the `whenThinking` policy OpenCode reactivates for thinking-on requests, #1071/#1484 — or `thinkingFormat: "zai"`); the `openai-completions` encoder surfaces those blocks via `reasoningContentField`, with a new branch for Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi, Xiaomi MiMo) that accept but don't require the field. Targets that can't replay unsigned reasoning (encrypted reasoning blobs, signed thought parts, non-reasoning models, thinking-disabled OpenCode) still text-demote so the reasoning survives as conversation context. ([#3437](https://github.com/can1357/oh-my-pi/pull/3437), [#3439](https://github.com/can1357/oh-my-pi/pull/3439) by [@roboomp](https://github.com/roboomp); [#3433](https://github.com/can1357/oh-my-pi/issues/3433), [#3434](https://github.com/can1357/oh-my-pi/issues/3434))
|
||||
- Fixed Bedrock cross-region inference profiles routing to `us-east-1` regardless of their geo prefix: a profile such as `eu.anthropic.claude-…` (or `apac.`/`au.`/`jp.`) sent to the hardcoded `us-east-1` endpoint returned HTTP 400 `The provided model identifier is invalid`. `streamBedrock` now derives the runtime region from the profile's geo prefix — honoring an ambient `AWS_REGION`/`AWS_DEFAULT_REGION` only when it can serve that geo and falling back to the geo's default region otherwise — while explicit per-request and ARN-embedded regions still win and region-agnostic `global.` profiles stay unchanged.
|
||||
- Fixed malformed tool calls (empty `name`) wedging entire sessions in HTTP 400 loops: when a model occasionally emits `{ "name": "", "arguments": "{}" }` (observed: GLM-5.2 + thinking on long turns), the agent rejected the call at execution time with `Tool not found`, but the malformed block plus its error `toolResult` stayed in conversation history and every subsequent request 400'd on `tool_use.name`/`tool_calls[i].function.name` validation until the user ran `/clear`. `transformMessages` — the canonical sanitize boundary every provider passes through — now drops `toolCall` blocks with empty/whitespace `name`, pairs them with their `toolResult` messages only inside the same assistant→tool-result window (per-id FIFO queue cleared at non-result boundaries, so stale malformed calls without a result cannot consume later valid duplicate-id outputs), and drops the assistant turn when it has no replayable content left. Defensive (provider-agnostic, fires regardless of model), idempotent (no-op on a clean history), and self-healing (one round-trip after the fix lands sanitizes an already-poisoned session). ([#3458](https://github.com/can1357/oh-my-pi/issues/3458))
|
||||
|
||||
@@ -192,14 +192,18 @@ function toPlainContent(
|
||||
};
|
||||
}
|
||||
|
||||
function convertMessage(message: Message, supportsImages: boolean): OllamaMessage {
|
||||
function convertMessage(
|
||||
message: Message,
|
||||
supportsImages: boolean,
|
||||
developerRole: "system" | "user" = "user",
|
||||
): OllamaMessage {
|
||||
if (message.role === "user") {
|
||||
const converted = toPlainContent(message.content, supportsImages);
|
||||
return { role: "user", ...converted };
|
||||
}
|
||||
if (message.role === "developer") {
|
||||
const converted = toPlainContent(message.content, supportsImages);
|
||||
return { role: "system", ...converted };
|
||||
return { role: developerRole, ...converted };
|
||||
}
|
||||
if (message.role === "toolResult") {
|
||||
const converted = toPlainContent(message.content, supportsImages);
|
||||
@@ -240,23 +244,27 @@ function convertMessage(message: Message, supportsImages: boolean): OllamaMessag
|
||||
}
|
||||
|
||||
function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaMessage[] {
|
||||
const messages: Message[] = [];
|
||||
// Emit one developer message per ordered system prompt. The wire role is mapped to "system"
|
||||
// by `convertMessage`, but keeping the prompts separate preserves prefix-cache stability:
|
||||
// if only the trailing prompt changes between calls, the leading system messages keep
|
||||
// their identical token prefix so KV-cache reuse covers them.
|
||||
for (const systemPrompt of normalizeSystemPrompts(context.systemPrompt)) {
|
||||
messages.push({
|
||||
role: "developer",
|
||||
content: systemPrompt,
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
}
|
||||
messages.push(...context.messages);
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
const systemMessages: Message[] = systemPrompts.map(systemPrompt => ({
|
||||
role: "developer",
|
||||
content: systemPrompt,
|
||||
timestamp: Date.now(),
|
||||
}));
|
||||
const messages: Message[] = [...systemMessages, ...context.messages];
|
||||
const isCloud = model.provider === "ollama-cloud";
|
||||
const supportsImages = model.input.includes("image");
|
||||
return transformMessages(messages, model).map(msg => {
|
||||
const converted = convertMessage(msg, supportsImages);
|
||||
return transformMessages(messages, model).map((msg, index) => {
|
||||
// Real `systemPrompt` entries (always emitted first) stay on Ollama's
|
||||
// `system` role. After the static prefix, a developer turn keeps `system`
|
||||
// when it's an agent-owned control instruction (empty/unexpected-stop
|
||||
// retries, checkpoint rewind warning, todo reminders — all carry
|
||||
// `attribution: "agent"`), but a user-attributed developer turn (auto-learn
|
||||
// capture nudge, advisor cards, file-mention companions) drops to `user`.
|
||||
// That keeps the in-conversation byte prefix stable for prefix caches
|
||||
// (llama.cpp, #3456) without demoting mandatory agent reminders.
|
||||
const developerRole =
|
||||
msg.role === "developer" && (index < systemPrompts.length || msg.attribution !== "user") ? "system" : "user";
|
||||
const converted = convertMessage(msg, supportsImages, developerRole);
|
||||
// Ollama cloud rejects requests when assistant history messages contain the `thinking`
|
||||
// field — it's valid in model responses but not accepted as a history input. Strip it
|
||||
// to prevent HTTP 400 errors. Local Ollama instances are unaffected.
|
||||
|
||||
@@ -70,6 +70,95 @@ describe("Ollama chat thinking controls", () => {
|
||||
|
||||
expect(payload?.think).toBe(false);
|
||||
});
|
||||
it("sends mid-conversation developer messages as user turns for llama.cpp cache reuse", async () => {
|
||||
let payload: OllamaChatRequestPayload | undefined;
|
||||
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const parsed: unknown = JSON.parse(String(init?.body));
|
||||
if (!isOllamaChatRequestPayload(parsed)) {
|
||||
throw new Error("Expected Ollama payload object");
|
||||
}
|
||||
payload = parsed;
|
||||
return new Response('{"message":{"content":"captured"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
|
||||
status: 200,
|
||||
});
|
||||
};
|
||||
const now = Date.now();
|
||||
const context: Context = {
|
||||
systemPrompt: ["static system"],
|
||||
messages: [
|
||||
{ role: "user", content: "Do work", timestamp: now - 2 },
|
||||
{
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "Done" }],
|
||||
api: "ollama-chat",
|
||||
provider: "ollama",
|
||||
model: "llama",
|
||||
usage: emptyUsage,
|
||||
stopReason: "stop",
|
||||
timestamp: now - 1,
|
||||
},
|
||||
{
|
||||
role: "developer",
|
||||
content: [{ type: "text", text: "Capture reusable lessons." }],
|
||||
attribution: "user",
|
||||
timestamp: now,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
await streamOllama(createReasoningOllamaModel(), context, {
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "user"]);
|
||||
expect(payload?.messages?.at(-1)?.content).toBe("Capture reusable lessons.");
|
||||
});
|
||||
|
||||
it("keeps agent-attributed developer reminders on the system role", async () => {
|
||||
let payload: OllamaChatRequestPayload | undefined;
|
||||
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const parsed: unknown = JSON.parse(String(init?.body));
|
||||
if (!isOllamaChatRequestPayload(parsed)) {
|
||||
throw new Error("Expected Ollama payload object");
|
||||
}
|
||||
payload = parsed;
|
||||
return new Response('{"message":{"content":"resumed"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
|
||||
status: 200,
|
||||
});
|
||||
};
|
||||
const now = Date.now();
|
||||
const context: Context = {
|
||||
systemPrompt: ["static system"],
|
||||
messages: [
|
||||
{ role: "user", content: "Do work", timestamp: now - 2 },
|
||||
{
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "Done" }],
|
||||
api: "ollama-chat",
|
||||
provider: "ollama",
|
||||
model: "llama",
|
||||
usage: emptyUsage,
|
||||
stopReason: "stop",
|
||||
timestamp: now - 1,
|
||||
},
|
||||
{
|
||||
role: "developer",
|
||||
content: [{ type: "text", text: "<system-warning>complete the checkpoint</system-warning>" }],
|
||||
attribution: "agent",
|
||||
timestamp: now,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
await streamOllama(createReasoningOllamaModel(), context, {
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(payload?.messages?.map(message => message.role)).toEqual(["system", "user", "assistant", "system"]);
|
||||
expect(payload?.messages?.at(-1)?.content).toContain("complete the checkpoint");
|
||||
});
|
||||
|
||||
it("omits tool-result images for text-only Ollama chat models", async () => {
|
||||
let payload: OllamaChatRequestPayload | undefined;
|
||||
|
||||
Reference in New Issue
Block a user