fix(ai): replay reasoning_content for DeepSeek V4 across compat hosts

DeepSeek V4 (v4-pro / v4-flash and reasoning-capable v3.x variants) reject follow-up
requests with 'The reasoning content in the thinking mode must be passed back to the
API' whenever a prior assistant tool-call turn lacks reasoning_content. The compat
flag that triggers placeholder injection was scoped to Kimi and OpenRouter-routed
reasoning models, so DeepSeek reached through api.deepseek.com, Deepinfra, Kilo,
NVIDIA NIM or Zenmux silently failed.

- Match DeepSeek family by provider, baseUrl, model id, or model name in
  detectOpenAICompat (gated by model.reasoning).
- Recompute hasReasoningField after injecting the placeholder so the existing
  'content === null && hasReasoningField -> ""' DeepSeek normalization actually
  runs on tool-only assistant turns.

Fixes #883
Fixes #810
This commit is contained in:
can1357
2026-04-30 04:51:23 +02:00
parent f5476d25b4
commit 784a457a88
3 changed files with 141 additions and 5 deletions
@@ -55,6 +55,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
const isKimiModel = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
const isQwen = model.id.toLowerCase().includes("qwen");
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
// upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra,
// Kilo, NVIDIA NIM, Zenmux, OpenRouter, …), so we match by model id/name as well as by
// provider/baseUrl. The flag is gated by `model.reasoning` because the invariant only
// applies when thinking mode is actually engaged.
const lowerId = model.id.toLowerCase();
const lowerName = (model.name ?? "").toLowerCase();
const isDeepseekFamily =
provider === "deepseek" ||
baseUrl.includes("deepseek.com") ||
lowerId.includes("deepseek") ||
lowerName.includes("deepseek");
const isNonStandard =
isCerebras ||
@@ -119,7 +132,9 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
// redacted/encrypted reasoning into DeepSeek's plaintext form, so cross-provider continuations
// rely on a placeholder — see `convertMessages` for the placeholder injection.
requiresReasoningContentForToolCalls:
isKimiModel || ((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
isKimiModel ||
(isDeepseekFamily && Boolean(model.reasoning)) ||
((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
requiresAssistantContentForToolCalls: isKimiModel,
openRouterRouting: undefined,
vercelGatewayRouting: undefined,
@@ -1225,10 +1225,6 @@ export function convertMessages(
}
const toolCalls = msg.content.filter(b => b.type === "toolCall") as ToolCall[];
const hasReasoningField =
(assistantMsg as any).reasoning_content !== undefined ||
(assistantMsg as any).reasoning !== undefined ||
(assistantMsg as any).reasoning_text !== undefined;
// Inject a `reasoning_content` placeholder on assistant tool-call turns when the backend
// rejects history without it. The compat flag captures the rule:
// - Kimi (native or via OpenCode-Go): chat completion endpoint demands the field.
@@ -1243,9 +1239,14 @@ export function convertMessages(
const stubsReasoningContent =
compat.requiresReasoningContentForToolCalls &&
(compat.thinkingFormat === "openai" || compat.thinkingFormat === "openrouter");
let hasReasoningField =
(assistantMsg as any).reasoning_content !== undefined ||
(assistantMsg as any).reasoning !== undefined ||
(assistantMsg as any).reasoning_text !== undefined;
if (toolCalls.length > 0 && stubsReasoningContent && !hasReasoningField) {
const reasoningField = compat.reasoningContentField ?? "reasoning_content";
(assistantMsg as any)[reasoningField] = ".";
hasReasoningField = true;
}
if (toolCalls.length > 0) {
assistantMsg.tool_calls = toolCalls.map((tc, toolCallIndex) => {
+120
View File
@@ -0,0 +1,120 @@
import { describe, expect, it } from "bun:test";
import { getBundledModel } from "../src/models";
import { convertMessages, detectCompat } from "../src/providers/openai-completions";
import type { AssistantMessage, Model } from "../src/types";
function deepseekModel(overrides: Partial<Model<"openai-completions">>): Model<"openai-completions"> {
return {
...getBundledModel("openai", "gpt-4o-mini"),
api: "openai-completions",
reasoning: true,
...overrides,
};
}
function assistantWithToolCall(model: Model<"openai-completions">): AssistantMessage {
return {
role: "assistant",
content: [
{ type: "text", text: "Calling a tool." },
{
type: "toolCall",
id: "call_repro_1",
name: "list_files",
arguments: { path: "." },
},
],
api: model.api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "toolUse",
timestamp: Date.now(),
};
}
describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay", () => {
it("flags requiresReasoningContentForToolCalls for deepseek-v4-pro on the official endpoint", () => {
const compat = detectCompat(
deepseekModel({
provider: "deepseek",
baseUrl: "https://api.deepseek.com/v1",
id: "deepseek-v4-pro",
}),
);
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
});
it("flags requiresReasoningContentForToolCalls for deepseek-v4 served by a non-deepseek host (e.g. Deepinfra)", () => {
const compat = detectCompat(
deepseekModel({
provider: "deepinfra",
baseUrl: "https://api.deepinfra.com/v1/openai",
id: "deepseek-ai/DeepSeek-V4-Flash",
}),
);
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
});
it("injects reasoning_content placeholder on assistant tool-call turn for deepseek-v4-pro", () => {
const model = deepseekModel({
provider: "deepseek",
baseUrl: "https://api.deepseek.com/v1",
id: "deepseek-v4-pro",
});
const compat = detectCompat(model);
const messages = convertMessages(model, { messages: [assistantWithToolCall(model)] }, compat);
const assistant = messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
expect(typeof reasoningContent).toBe("string");
expect((reasoningContent as string).length).toBeGreaterThan(0);
});
it("normalizes assistant content to '' when reasoning_content placeholder is injected (DeepSeek invariant)", () => {
const model = deepseekModel({
provider: "deepinfra",
baseUrl: "https://api.deepinfra.com/v1/openai",
id: "deepseek-ai/DeepSeek-V4-Pro",
});
const compat = detectCompat(model);
// Assistant turn whose only content is a tool call (no text) - matches what the SDK
// produces after a pure tool-use turn. content must end up "" (not null) because
// DeepSeek rejects null content alongside reasoning_content.
const toolOnly: AssistantMessage = {
role: "assistant",
content: [
{
type: "toolCall",
id: "call_repro_2",
name: "list_files",
arguments: { path: "." },
},
],
api: model.api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "toolUse",
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [toolOnly] }, compat);
const assistant = messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect((assistant as { content: unknown }).content).toBe("");
});
});