fix(ai): fixed DeepSeek V4 direct API reasoning payload handling

- Updated DeepSeek V4 direct API compat to map unsupported low reasoning levels to high, use `max_tokens`, set `thinking` mode, and suppress `tool_choice` on tool-call requests.
- Adjusted OpenAI compat detection and resolution to specially handle direct DeepSeek reasoning flows while preserving OpenRouter behavior and merging model-provided compat overrides.
- Added regression coverage for issue #1207 and updated DeepSeek model tests to assert the new reasoning mapping, payload shape, and tool-call requirements.
This commit is contained in:
can1357
2026-05-20 15:46:25 +09:00
parent ea1a508217
commit 7f6625e13a
6 changed files with 176 additions and 23 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed DeepSeek V4 direct API requests with tools to keep documented thinking mode instead of dropping reasoning: lower OMP efforts now map to DeepSeek's supported `high`, `tool_choice` is omitted, `thinking: { type: "enabled" }` and `max_tokens` are sent, and partial user `reasoningEffortMap` overrides merge with DeepSeek defaults. ([#1207](https://github.com/can1357/oh-my-pi/issues/1207))
## [15.1.7] - 2026-05-19
### Added
@@ -2083,18 +2083,26 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
// ids are kept off the catalog until the issue thread asks for them.
filterModel: (id, m) => m.tool_call === true && id.startsWith("deepseek-v4"),
compat: {
// xhigh maps to DeepSeek's `max` reasoning_effort (#830 thread).
// DeepSeek V4 only accepts `high`/`max`; map lower OMP levels upward so
// subagent "minimal" turns stay in documented thinking mode instead of
// sending unsupported effort strings.
supportsDeveloperRole: false,
supportsReasoningEffort: true,
reasoningEffortMap: { xhigh: "max" },
// `tool_choice` returns 400 against DeepSeek when reasoning_effort is set
// (per the issue thread). Tool calls still work without the parameter.
reasoningEffortMap: { minimal: "high", low: "high", medium: "high", high: "high", xhigh: "max" },
maxTokensField: "max_tokens",
// DeepSeek V4 thinking mode rejects the `tool_choice` control parameter.
// Tool calls still work without it; the API defaults to auto when tools exist.
supportsToolChoice: false,
// DeepSeek V4's OpenAI format docs enable thinking with both the toggle and
// reasoning_effort. Keep the toggle explicit for built-in models.
extraBody: { thinking: { type: "enabled" } },
// DeepSeek emits chain-of-thought via `reasoning_content` and requires it
// to round-trip on assistant tool-call messages so the model can resume
// from prior thinking (interleaved.field=reasoning_content on models.dev,
// matches the kimi/openrouter handling already in detectCompat).
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: true,
requiresAssistantContentForToolCalls: true,
},
}),
];
@@ -79,7 +79,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
baseUrl.includes("deepseek.com") ||
lowerId.includes("deepseek") ||
lowerName.includes("deepseek");
const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com");
const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning);
const isNonStandard =
isCerebras ||
provider === "xai" ||
@@ -102,7 +103,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
provider === "mistral" ||
baseUrl.includes("mistral.ai") ||
baseUrl.includes("chutes.ai") ||
baseUrl.includes("fireworks.ai");
baseUrl.includes("fireworks.ai") ||
isDirectDeepseekApi;
const isGrok = provider === "xai" || baseUrl.includes("api.x.ai");
const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai");
@@ -162,7 +164,13 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
xhigh: "default",
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
: isDeepseekFamily && model.reasoning
? { xhigh: "max" }
? ({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
: {};
return {
@@ -173,8 +181,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
reasoningEffortMap,
supportsUsageInStreaming: !isCerebras,
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning),
supportsToolChoice: true,
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
supportsToolChoice: !isDirectDeepseekReasoning,
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
requiresToolResultName: isMistral,
requiresAssistantAfterToolResult: false,
@@ -204,11 +212,11 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
// DeepSeek V4 rejects synthetic reasoning_content placeholders (".") on tool-call turns.
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
allowsSyntheticReasoningContentForToolCalls: !isDeepseekFamily || !model.reasoning,
requiresAssistantContentForToolCalls: isKimiModel,
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
openRouterRouting: undefined,
vercelGatewayRouting: undefined,
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
extraBody: undefined,
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
toolStrictMode: isCerebras ? "all_strict" : "mixed",
};
}
@@ -235,7 +243,7 @@ export function resolveOpenAICompat(
supportsMultipleSystemMessages:
model.compat.supportsMultipleSystemMessages ?? detected.supportsMultipleSystemMessages,
supportsReasoningEffort: model.compat.supportsReasoningEffort ?? detected.supportsReasoningEffort,
reasoningEffortMap: model.compat.reasoningEffortMap ?? detected.reasoningEffortMap,
reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) },
supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming,
supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice,
maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField,
@@ -259,7 +267,7 @@ export function resolveOpenAICompat(
openRouterRouting: model.compat.openRouterRouting ?? detected.openRouterRouting,
vercelGatewayRouting: model.compat.vercelGatewayRouting ?? detected.vercelGatewayRouting,
supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode,
extraBody: model.compat.extraBody,
extraBody: model.compat.extraBody ?? detected.extraBody,
toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode,
};
}
@@ -47,7 +47,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
// Fix 1: reasoningEffortMap for DeepSeek-family on any provider
// ----------------------------------------------------------------
describe("reasoningEffortMap (Fix 1)", () => {
it("maps xhigh → max for DeepSeek-family on opencode-go", () => {
it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => {
const compat = detectCompat(
deepseekModel({
provider: "opencode-go",
@@ -55,10 +55,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
id: "deepseek-v4-flash",
}),
);
expect(compat.reasoningEffortMap.xhigh).toBe("max");
expect(compat.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
});
});
it("maps xhigh → max for DeepSeek-family on NVIDIA", () => {
it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => {
const compat = detectCompat(
deepseekModel({
provider: "nvidia",
@@ -66,10 +72,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
id: "deepseek-ai/deepseek-v4-flash",
}),
);
expect(compat.reasoningEffortMap.xhigh).toBe("max");
expect(compat.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
});
});
it("maps xhigh → max for DeepSeek on the official endpoint", () => {
it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => {
const compat = detectCompat(
deepseekModel({
provider: "deepseek",
@@ -77,7 +89,13 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
id: "deepseek-v4-pro",
}),
);
expect(compat.reasoningEffortMap.xhigh).toBe("max");
expect(compat.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
});
});
it("does NOT map xhigh for non-DeepSeek models", () => {
+105
View File
@@ -0,0 +1,105 @@
import { describe, expect, it } from "bun:test";
import { getBundledModel } from "@oh-my-pi/pi-ai/models";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat";
import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types";
import * as z from "zod/v4";
const echoTool: Tool = {
name: "echo",
description: "Echo input",
parameters: z.object({ text: z.string() }),
};
const contextWithTools: Context = {
messages: [{ role: "user", content: "call echo", timestamp: Date.now() }],
tools: [echoTool],
};
function abortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
async function capturePayload(model: Model<"openai-completions">): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, contextWithTools, {
apiKey: "test-key",
signal: abortedSignal(),
reasoning: "minimal",
toolChoice: "auto",
maxTokens: 123,
onPayload: payload => resolve(payload),
});
return (await promise) as Record<string, unknown>;
}
function customDeepseekFlash(): Model<"openai-completions"> {
return {
...getBundledModel("openai", "gpt-4o-mini"),
api: "openai-completions",
id: "deepseek-v4-flash",
name: "DeepSeek V4 Flash",
provider: "ds",
baseUrl: "https://api.deepseek.com/v1",
reasoning: true,
compat: {
supportsReasoningEffort: true,
reasoningEffortMap: { xhigh: "max" },
},
};
}
describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => {
it("detects the documented direct DeepSeek V4 compat shape", () => {
const model = getBundledModel("deepseek", "deepseek-v4-flash") as Model<"openai-completions">;
const compat = detectOpenAICompat(model);
expect(compat.supportsToolChoice).toBe(false);
expect(compat.maxTokensField).toBe("max_tokens");
expect(compat.extraBody).toEqual({ thinking: { type: "enabled" } });
expect(compat.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
});
});
it("merges partial user reasoning maps with DeepSeek defaults", () => {
const compat = resolveOpenAICompat(customDeepseekFlash());
expect(compat.supportsToolChoice).toBe(false);
expect(compat.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
xhigh: "max",
});
});
it("omits tool_choice but preserves documented reasoning when tools are present", async () => {
const body = await capturePayload(customDeepseekFlash());
expect(body.tools).toBeDefined();
expect(body.tool_choice).toBeUndefined();
expect(body.reasoning_effort).toBe("high");
expect(body.thinking).toEqual({ type: "enabled" });
expect(body.max_tokens).toBe(123);
expect(body.max_completion_tokens).toBeUndefined();
});
it("preserves OpenRouter reasoning when tool_choice auto is present", async () => {
const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">;
const compat = detectOpenAICompat(model);
const body = await capturePayload(model);
expect(compat.disableReasoningOnToolChoice).toBe(false);
expect(body.tools).toBeDefined();
expect(body.tool_choice).toBe("auto");
expect(body.reasoning).toEqual({ effort: "high" });
expect(body.reasoning_effort).toBeUndefined();
});
});
+14 -4
View File
@@ -33,15 +33,25 @@ describe("deepseek built-in provider (issue #830)", () => {
expect(descriptor?.modelsDevKey).toBe("deepseek");
expect(descriptor?.api).toBe("openai-completions");
expect(descriptor?.baseUrl).toBe("https://api.deepseek.com");
// Per-model compat: deepseek-v4 reasoning models leak chat-template tool-call markers
// (#798) and 400 on tool_choice when xhigh effort is used (#830 thread). Reasoning content
// must round-trip on tool calls (interleaved.field=reasoning_content from models.dev).
// Per-model compat: DeepSeek V4 supports thinking-mode tool calls, but only
// with `high`/`max` effort, no explicit `tool_choice`, max_tokens, and
// reasoning_content replay.
const compat =
descriptor?.api === "openai-completions" ? (descriptor.compat as OpenAICompat | undefined) : undefined;
expect(compat?.supportsDeveloperRole).toBe(false);
expect(compat?.supportsReasoningEffort).toBe(true);
expect(compat?.supportsToolChoice).toBe(false);
expect(compat?.maxTokensField).toBe("max_tokens");
expect(compat?.requiresReasoningContentForToolCalls).toBe(true);
expect(compat?.requiresAssistantContentForToolCalls).toBe(true);
expect(compat?.reasoningContentField).toBe("reasoning_content");
expect(compat?.reasoningEffortMap?.xhigh).toBe("max");
expect(compat?.extraBody).toEqual({ thinking: { type: "enabled" } });
expect(compat?.reasoningEffortMap).toMatchObject({
minimal: "high",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
});
});
});