Merge PR #1000: fix(ai): strip thinking field from history messages on ollama-cloud

Ollama Cloud rejects the thinking field in assistant history messages
(live streaming is fine), breaking multi-turn conversations after the
first reply. Strips it in convertMessages on the cloud path only;
local Ollama uses openai-responses and is unaffected.

Closes #1000
This commit is contained in:
can1357
2026-05-13 02:25:43 +02:00
2 changed files with 71 additions and 1 deletions
+12 -1
View File
@@ -199,7 +199,18 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM
});
}
messages.push(...context.messages);
return transformMessages(messages, model).map(convertMessage);
const isCloud = model.provider === "ollama-cloud";
return transformMessages(messages, model).map(msg => {
const converted = convertMessage(msg);
// Ollama cloud rejects requests when assistant history messages contain the `thinking`
// field — it's valid in model responses but not accepted as a history input. Strip it
// to prevent HTTP 400 errors. Local Ollama instances are unaffected.
if (isCloud && converted.role === "assistant" && converted.thinking) {
const { thinking: _t, ...rest } = converted;
return rest;
}
return converted;
});
}
function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefined {
@@ -415,6 +415,65 @@ describe("ollama-cloud provider support", () => {
});
});
test("strips `thinking` from assistant history messages on ollama-cloud", async () => {
let requestBody: Record<string, unknown> | undefined;
global.fetch = vi.fn(async (_input, init) => {
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
return createNdjsonResponse([
{ model: "gpt-oss:120b", message: { role: "assistant", content: "ok" }, done: false },
{ model: "gpt-oss:120b", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
]);
}) as unknown as typeof fetch;
const context: Context = {
messages: [
{ role: "user", content: "kick off", timestamp: Date.now() },
{
role: "assistant",
content: [
{ type: "thinking", thinking: "internal reasoning" },
{ type: "toolCall", id: "tool-1", name: "read_file", arguments: { path: "README.md" } },
],
api: "ollama-chat",
provider: "ollama-cloud",
model: "gpt-oss:120b",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "toolUse",
timestamp: Date.now(),
},
{
role: "toolResult",
toolCallId: "tool-1",
toolName: "read_file",
content: [{ type: "text", text: "README contents" }],
isError: false,
timestamp: Date.now(),
},
],
tools: [readFileTool],
};
await stream(cloudModel, context, { apiKey: "cloud-test-key" }).result();
const messages = requestBody?.messages as Array<Record<string, unknown>> | undefined;
const assistant = messages?.find(message => message.role === "assistant");
expect(assistant).toBeDefined();
expect(assistant).not.toHaveProperty("thinking");
expect(assistant?.tool_calls).toEqual([
{
type: "function",
function: { name: "read_file", arguments: { path: "README.md" } },
},
]);
});
test("emits one Ollama system message per ordered system prompt entry", async () => {
let requestBody: Record<string, unknown> | undefined;
global.fetch = vi.fn(async (_input, init) => {