fix(ai): fixed DeepSeek V4 direct API reasoning payload handling
- Updated DeepSeek V4 direct API compat to map unsupported low reasoning levels to high, use `max_tokens`, set `thinking` mode, and suppress `tool_choice` on tool-call requests. - Adjusted OpenAI compat detection and resolution to specially handle direct DeepSeek reasoning flows while preserving OpenRouter behavior and merging model-provided compat overrides. - Added regression coverage for issue #1207 and updated DeepSeek model tests to assert the new reasoning mapping, payload shape, and tool-call requirements.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed DeepSeek V4 direct API requests with tools to keep documented thinking mode instead of dropping reasoning: lower OMP efforts now map to DeepSeek's supported `high`, `tool_choice` is omitted, `thinking: { type: "enabled" }` and `max_tokens` are sent, and partial user `reasoningEffortMap` overrides merge with DeepSeek defaults. ([#1207](https://github.com/can1357/oh-my-pi/issues/1207))
|
||||
|
||||
## [15.1.7] - 2026-05-19
|
||||
### Added
|
||||
|
||||
|
||||
@@ -2083,18 +2083,26 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
||||
// ids are kept off the catalog until the issue thread asks for them.
|
||||
filterModel: (id, m) => m.tool_call === true && id.startsWith("deepseek-v4"),
|
||||
compat: {
|
||||
// xhigh maps to DeepSeek's `max` reasoning_effort (#830 thread).
|
||||
// DeepSeek V4 only accepts `high`/`max`; map lower OMP levels upward so
|
||||
// subagent "minimal" turns stay in documented thinking mode instead of
|
||||
// sending unsupported effort strings.
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortMap: { xhigh: "max" },
|
||||
// `tool_choice` returns 400 against DeepSeek when reasoning_effort is set
|
||||
// (per the issue thread). Tool calls still work without the parameter.
|
||||
reasoningEffortMap: { minimal: "high", low: "high", medium: "high", high: "high", xhigh: "max" },
|
||||
maxTokensField: "max_tokens",
|
||||
// DeepSeek V4 thinking mode rejects the `tool_choice` control parameter.
|
||||
// Tool calls still work without it; the API defaults to auto when tools exist.
|
||||
supportsToolChoice: false,
|
||||
// DeepSeek V4's OpenAI format docs enable thinking with both the toggle and
|
||||
// reasoning_effort. Keep the toggle explicit for built-in models.
|
||||
extraBody: { thinking: { type: "enabled" } },
|
||||
// DeepSeek emits chain-of-thought via `reasoning_content` and requires it
|
||||
// to round-trip on assistant tool-call messages so the model can resume
|
||||
// from prior thinking (interleaved.field=reasoning_content on models.dev,
|
||||
// matches the kimi/openrouter handling already in detectCompat).
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: true,
|
||||
requiresAssistantContentForToolCalls: true,
|
||||
},
|
||||
}),
|
||||
];
|
||||
|
||||
@@ -79,7 +79,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
baseUrl.includes("deepseek.com") ||
|
||||
lowerId.includes("deepseek") ||
|
||||
lowerName.includes("deepseek");
|
||||
|
||||
const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com");
|
||||
const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning);
|
||||
const isNonStandard =
|
||||
isCerebras ||
|
||||
provider === "xai" ||
|
||||
@@ -102,7 +103,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
provider === "mistral" ||
|
||||
baseUrl.includes("mistral.ai") ||
|
||||
baseUrl.includes("chutes.ai") ||
|
||||
baseUrl.includes("fireworks.ai");
|
||||
baseUrl.includes("fireworks.ai") ||
|
||||
isDirectDeepseekApi;
|
||||
const isGrok = provider === "xai" || baseUrl.includes("api.x.ai");
|
||||
const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai");
|
||||
|
||||
@@ -162,7 +164,13 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
xhigh: "default",
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: isDeepseekFamily && model.reasoning
|
||||
? { xhigh: "max" }
|
||||
? ({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: {};
|
||||
|
||||
return {
|
||||
@@ -173,8 +181,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
reasoningEffortMap,
|
||||
supportsUsageInStreaming: !isCerebras,
|
||||
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
|
||||
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning),
|
||||
supportsToolChoice: true,
|
||||
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
|
||||
supportsToolChoice: !isDirectDeepseekReasoning,
|
||||
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
|
||||
requiresToolResultName: isMistral,
|
||||
requiresAssistantAfterToolResult: false,
|
||||
@@ -204,11 +212,11 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
// DeepSeek V4 rejects synthetic reasoning_content placeholders (".") on tool-call turns.
|
||||
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
|
||||
allowsSyntheticReasoningContentForToolCalls: !isDeepseekFamily || !model.reasoning,
|
||||
requiresAssistantContentForToolCalls: isKimiModel,
|
||||
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
|
||||
openRouterRouting: undefined,
|
||||
vercelGatewayRouting: undefined,
|
||||
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
|
||||
extraBody: undefined,
|
||||
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
|
||||
toolStrictMode: isCerebras ? "all_strict" : "mixed",
|
||||
};
|
||||
}
|
||||
@@ -235,7 +243,7 @@ export function resolveOpenAICompat(
|
||||
supportsMultipleSystemMessages:
|
||||
model.compat.supportsMultipleSystemMessages ?? detected.supportsMultipleSystemMessages,
|
||||
supportsReasoningEffort: model.compat.supportsReasoningEffort ?? detected.supportsReasoningEffort,
|
||||
reasoningEffortMap: model.compat.reasoningEffortMap ?? detected.reasoningEffortMap,
|
||||
reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) },
|
||||
supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming,
|
||||
supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice,
|
||||
maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField,
|
||||
@@ -259,7 +267,7 @@ export function resolveOpenAICompat(
|
||||
openRouterRouting: model.compat.openRouterRouting ?? detected.openRouterRouting,
|
||||
vercelGatewayRouting: model.compat.vercelGatewayRouting ?? detected.vercelGatewayRouting,
|
||||
supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode,
|
||||
extraBody: model.compat.extraBody,
|
||||
extraBody: model.compat.extraBody ?? detected.extraBody,
|
||||
toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -47,7 +47,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
// Fix 1: reasoningEffortMap for DeepSeek-family on any provider
|
||||
// ----------------------------------------------------------------
|
||||
describe("reasoningEffortMap (Fix 1)", () => {
|
||||
it("maps xhigh → max for DeepSeek-family on opencode-go", () => {
|
||||
it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => {
|
||||
const compat = detectCompat(
|
||||
deepseekModel({
|
||||
provider: "opencode-go",
|
||||
@@ -55,10 +55,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
id: "deepseek-v4-flash",
|
||||
}),
|
||||
);
|
||||
expect(compat.reasoningEffortMap.xhigh).toBe("max");
|
||||
expect(compat.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
|
||||
it("maps xhigh → max for DeepSeek-family on NVIDIA", () => {
|
||||
it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => {
|
||||
const compat = detectCompat(
|
||||
deepseekModel({
|
||||
provider: "nvidia",
|
||||
@@ -66,10 +72,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
id: "deepseek-ai/deepseek-v4-flash",
|
||||
}),
|
||||
);
|
||||
expect(compat.reasoningEffortMap.xhigh).toBe("max");
|
||||
expect(compat.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
|
||||
it("maps xhigh → max for DeepSeek on the official endpoint", () => {
|
||||
it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => {
|
||||
const compat = detectCompat(
|
||||
deepseekModel({
|
||||
provider: "deepseek",
|
||||
@@ -77,7 +89,13 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
id: "deepseek-v4-pro",
|
||||
}),
|
||||
);
|
||||
expect(compat.reasoningEffortMap.xhigh).toBe("max");
|
||||
expect(compat.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
|
||||
it("does NOT map xhigh for non-DeepSeek models", () => {
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-ai/models";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat";
|
||||
import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types";
|
||||
import * as z from "zod/v4";
|
||||
|
||||
const echoTool: Tool = {
|
||||
name: "echo",
|
||||
description: "Echo input",
|
||||
parameters: z.object({ text: z.string() }),
|
||||
};
|
||||
|
||||
const contextWithTools: Context = {
|
||||
messages: [{ role: "user", content: "call echo", timestamp: Date.now() }],
|
||||
tools: [echoTool],
|
||||
};
|
||||
|
||||
function abortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
async function capturePayload(model: Model<"openai-completions">): Promise<Record<string, unknown>> {
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, contextWithTools, {
|
||||
apiKey: "test-key",
|
||||
signal: abortedSignal(),
|
||||
reasoning: "minimal",
|
||||
toolChoice: "auto",
|
||||
maxTokens: 123,
|
||||
onPayload: payload => resolve(payload),
|
||||
});
|
||||
return (await promise) as Record<string, unknown>;
|
||||
}
|
||||
|
||||
function customDeepseekFlash(): Model<"openai-completions"> {
|
||||
return {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
api: "openai-completions",
|
||||
id: "deepseek-v4-flash",
|
||||
name: "DeepSeek V4 Flash",
|
||||
provider: "ds",
|
||||
baseUrl: "https://api.deepseek.com/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortMap: { xhigh: "max" },
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => {
|
||||
it("detects the documented direct DeepSeek V4 compat shape", () => {
|
||||
const model = getBundledModel("deepseek", "deepseek-v4-flash") as Model<"openai-completions">;
|
||||
const compat = detectOpenAICompat(model);
|
||||
|
||||
expect(compat.supportsToolChoice).toBe(false);
|
||||
expect(compat.maxTokensField).toBe("max_tokens");
|
||||
expect(compat.extraBody).toEqual({ thinking: { type: "enabled" } });
|
||||
expect(compat.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
|
||||
it("merges partial user reasoning maps with DeepSeek defaults", () => {
|
||||
const compat = resolveOpenAICompat(customDeepseekFlash());
|
||||
|
||||
expect(compat.supportsToolChoice).toBe(false);
|
||||
expect(compat.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
|
||||
it("omits tool_choice but preserves documented reasoning when tools are present", async () => {
|
||||
const body = await capturePayload(customDeepseekFlash());
|
||||
|
||||
expect(body.tools).toBeDefined();
|
||||
expect(body.tool_choice).toBeUndefined();
|
||||
expect(body.reasoning_effort).toBe("high");
|
||||
expect(body.thinking).toEqual({ type: "enabled" });
|
||||
expect(body.max_tokens).toBe(123);
|
||||
expect(body.max_completion_tokens).toBeUndefined();
|
||||
});
|
||||
|
||||
it("preserves OpenRouter reasoning when tool_choice auto is present", async () => {
|
||||
const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">;
|
||||
const compat = detectOpenAICompat(model);
|
||||
const body = await capturePayload(model);
|
||||
|
||||
expect(compat.disableReasoningOnToolChoice).toBe(false);
|
||||
expect(body.tools).toBeDefined();
|
||||
expect(body.tool_choice).toBe("auto");
|
||||
expect(body.reasoning).toEqual({ effort: "high" });
|
||||
expect(body.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -33,15 +33,25 @@ describe("deepseek built-in provider (issue #830)", () => {
|
||||
expect(descriptor?.modelsDevKey).toBe("deepseek");
|
||||
expect(descriptor?.api).toBe("openai-completions");
|
||||
expect(descriptor?.baseUrl).toBe("https://api.deepseek.com");
|
||||
// Per-model compat: deepseek-v4 reasoning models leak chat-template tool-call markers
|
||||
// (#798) and 400 on tool_choice when xhigh effort is used (#830 thread). Reasoning content
|
||||
// must round-trip on tool calls (interleaved.field=reasoning_content from models.dev).
|
||||
// Per-model compat: DeepSeek V4 supports thinking-mode tool calls, but only
|
||||
// with `high`/`max` effort, no explicit `tool_choice`, max_tokens, and
|
||||
// reasoning_content replay.
|
||||
const compat =
|
||||
descriptor?.api === "openai-completions" ? (descriptor.compat as OpenAICompat | undefined) : undefined;
|
||||
expect(compat?.supportsDeveloperRole).toBe(false);
|
||||
expect(compat?.supportsReasoningEffort).toBe(true);
|
||||
expect(compat?.supportsToolChoice).toBe(false);
|
||||
expect(compat?.maxTokensField).toBe("max_tokens");
|
||||
expect(compat?.requiresReasoningContentForToolCalls).toBe(true);
|
||||
expect(compat?.requiresAssistantContentForToolCalls).toBe(true);
|
||||
expect(compat?.reasoningContentField).toBe("reasoning_content");
|
||||
expect(compat?.reasoningEffortMap?.xhigh).toBe("max");
|
||||
expect(compat?.extraBody).toEqual({ thinking: { type: "enabled" } });
|
||||
expect(compat?.reasoningEffortMap).toMatchObject({
|
||||
minimal: "high",
|
||||
low: "high",
|
||||
medium: "high",
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user