fix(ai): replay reasoning_content for DeepSeek V4 across compat hosts
DeepSeek V4 (v4-pro / v4-flash and reasoning-capable v3.x variants) reject follow-up requests with 'The reasoning content in the thinking mode must be passed back to the API' whenever a prior assistant tool-call turn lacks reasoning_content. The compat flag that triggers placeholder injection was scoped to Kimi and OpenRouter-routed reasoning models, so DeepSeek reached through api.deepseek.com, Deepinfra, Kilo, NVIDIA NIM or Zenmux silently failed. - Match DeepSeek family by provider, baseUrl, model id, or model name in detectOpenAICompat (gated by model.reasoning). - Recompute hasReasoningField after injecting the placeholder so the existing 'content === null && hasReasoningField -> ""' DeepSeek normalization actually runs on tool-only assistant turns. Fixes #883 Fixes #810
This commit is contained in:
@@ -55,6 +55,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
const isKimiModel = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
|
||||
const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
|
||||
const isQwen = model.id.toLowerCase().includes("qwen");
|
||||
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
|
||||
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
|
||||
// upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra,
|
||||
// Kilo, NVIDIA NIM, Zenmux, OpenRouter, …), so we match by model id/name as well as by
|
||||
// provider/baseUrl. The flag is gated by `model.reasoning` because the invariant only
|
||||
// applies when thinking mode is actually engaged.
|
||||
const lowerId = model.id.toLowerCase();
|
||||
const lowerName = (model.name ?? "").toLowerCase();
|
||||
const isDeepseekFamily =
|
||||
provider === "deepseek" ||
|
||||
baseUrl.includes("deepseek.com") ||
|
||||
lowerId.includes("deepseek") ||
|
||||
lowerName.includes("deepseek");
|
||||
|
||||
const isNonStandard =
|
||||
isCerebras ||
|
||||
@@ -119,7 +132,9 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
// redacted/encrypted reasoning into DeepSeek's plaintext form, so cross-provider continuations
|
||||
// rely on a placeholder — see `convertMessages` for the placeholder injection.
|
||||
requiresReasoningContentForToolCalls:
|
||||
isKimiModel || ((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
|
||||
isKimiModel ||
|
||||
(isDeepseekFamily && Boolean(model.reasoning)) ||
|
||||
((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
|
||||
requiresAssistantContentForToolCalls: isKimiModel,
|
||||
openRouterRouting: undefined,
|
||||
vercelGatewayRouting: undefined,
|
||||
|
||||
@@ -1225,10 +1225,6 @@ export function convertMessages(
|
||||
}
|
||||
|
||||
const toolCalls = msg.content.filter(b => b.type === "toolCall") as ToolCall[];
|
||||
const hasReasoningField =
|
||||
(assistantMsg as any).reasoning_content !== undefined ||
|
||||
(assistantMsg as any).reasoning !== undefined ||
|
||||
(assistantMsg as any).reasoning_text !== undefined;
|
||||
// Inject a `reasoning_content` placeholder on assistant tool-call turns when the backend
|
||||
// rejects history without it. The compat flag captures the rule:
|
||||
// - Kimi (native or via OpenCode-Go): chat completion endpoint demands the field.
|
||||
@@ -1243,9 +1239,14 @@ export function convertMessages(
|
||||
const stubsReasoningContent =
|
||||
compat.requiresReasoningContentForToolCalls &&
|
||||
(compat.thinkingFormat === "openai" || compat.thinkingFormat === "openrouter");
|
||||
let hasReasoningField =
|
||||
(assistantMsg as any).reasoning_content !== undefined ||
|
||||
(assistantMsg as any).reasoning !== undefined ||
|
||||
(assistantMsg as any).reasoning_text !== undefined;
|
||||
if (toolCalls.length > 0 && stubsReasoningContent && !hasReasoningField) {
|
||||
const reasoningField = compat.reasoningContentField ?? "reasoning_content";
|
||||
(assistantMsg as any)[reasoningField] = ".";
|
||||
hasReasoningField = true;
|
||||
}
|
||||
if (toolCalls.length > 0) {
|
||||
assistantMsg.tool_calls = toolCalls.map((tc, toolCallIndex) => {
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { getBundledModel } from "../src/models";
|
||||
import { convertMessages, detectCompat } from "../src/providers/openai-completions";
|
||||
import type { AssistantMessage, Model } from "../src/types";
|
||||
|
||||
function deepseekModel(overrides: Partial<Model<"openai-completions">>): Model<"openai-completions"> {
|
||||
return {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
api: "openai-completions",
|
||||
reasoning: true,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function assistantWithToolCall(model: Model<"openai-completions">): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "text", text: "Calling a tool." },
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call_repro_1",
|
||||
name: "list_files",
|
||||
arguments: { path: "." },
|
||||
},
|
||||
],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay", () => {
|
||||
it("flags requiresReasoningContentForToolCalls for deepseek-v4-pro on the official endpoint", () => {
|
||||
const compat = detectCompat(
|
||||
deepseekModel({
|
||||
provider: "deepseek",
|
||||
baseUrl: "https://api.deepseek.com/v1",
|
||||
id: "deepseek-v4-pro",
|
||||
}),
|
||||
);
|
||||
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
|
||||
});
|
||||
|
||||
it("flags requiresReasoningContentForToolCalls for deepseek-v4 served by a non-deepseek host (e.g. Deepinfra)", () => {
|
||||
const compat = detectCompat(
|
||||
deepseekModel({
|
||||
provider: "deepinfra",
|
||||
baseUrl: "https://api.deepinfra.com/v1/openai",
|
||||
id: "deepseek-ai/DeepSeek-V4-Flash",
|
||||
}),
|
||||
);
|
||||
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
|
||||
});
|
||||
|
||||
it("injects reasoning_content placeholder on assistant tool-call turn for deepseek-v4-pro", () => {
|
||||
const model = deepseekModel({
|
||||
provider: "deepseek",
|
||||
baseUrl: "https://api.deepseek.com/v1",
|
||||
id: "deepseek-v4-pro",
|
||||
});
|
||||
const compat = detectCompat(model);
|
||||
const messages = convertMessages(model, { messages: [assistantWithToolCall(model)] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
|
||||
expect(typeof reasoningContent).toBe("string");
|
||||
expect((reasoningContent as string).length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("normalizes assistant content to '' when reasoning_content placeholder is injected (DeepSeek invariant)", () => {
|
||||
const model = deepseekModel({
|
||||
provider: "deepinfra",
|
||||
baseUrl: "https://api.deepinfra.com/v1/openai",
|
||||
id: "deepseek-ai/DeepSeek-V4-Pro",
|
||||
});
|
||||
const compat = detectCompat(model);
|
||||
// Assistant turn whose only content is a tool call (no text) - matches what the SDK
|
||||
// produces after a pure tool-use turn. content must end up "" (not null) because
|
||||
// DeepSeek rejects null content alongside reasoning_content.
|
||||
const toolOnly: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call_repro_2",
|
||||
name: "list_files",
|
||||
arguments: { path: "." },
|
||||
},
|
||||
],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [toolOnly] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect((assistant as { content: unknown }).content).toBe("");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user