From 7f6625e13abe9a255b5db65a4c42feb0be99eab2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 20 May 2026 15:46:25 +0900 Subject: [PATCH] fix(ai): fixed DeepSeek V4 direct API reasoning payload handling - Updated DeepSeek V4 direct API compat to map unsupported low reasoning levels to high, use `max_tokens`, set `thinking` mode, and suppress `tool_choice` on tool-call requests. - Adjusted OpenAI compat detection and resolution to specially handle direct DeepSeek reasoning flows while preserving OpenRouter behavior and merging model-provided compat overrides. - Added regression coverage for issue #1207 and updated DeepSeek model tests to assert the new reasoning mapping, payload shape, and tool-call requirements. --- packages/ai/CHANGELOG.md | 4 + .../ai/src/provider-models/openai-compat.ts | 16 ++- .../providers/openai-completions-compat.ts | 26 +++-- .../test/deepseek-reasoning-content.test.ts | 30 ++++- packages/ai/test/issue-1207-repro.test.ts | 105 ++++++++++++++++++ packages/ai/test/issue-830-repro.test.ts | 18 ++- 6 files changed, 176 insertions(+), 23 deletions(-) create mode 100644 packages/ai/test/issue-1207-repro.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 14ffec1d2..51a4fec7b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed DeepSeek V4 direct API requests with tools to keep documented thinking mode instead of dropping reasoning: lower OMP efforts now map to DeepSeek's supported `high`, `tool_choice` is omitted, `thinking: { type: "enabled" }` and `max_tokens` are sent, and partial user `reasoningEffortMap` overrides merge with DeepSeek defaults. ([#1207](https://github.com/can1357/oh-my-pi/issues/1207)) + ## [15.1.7] - 2026-05-19 ### Added diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index b56e3265c..5f29b1571 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -2083,18 +2083,26 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor // ids are kept off the catalog until the issue thread asks for them. filterModel: (id, m) => m.tool_call === true && id.startsWith("deepseek-v4"), compat: { - // xhigh maps to DeepSeek's `max` reasoning_effort (#830 thread). + // DeepSeek V4 only accepts `high`/`max`; map lower OMP levels upward so + // subagent "minimal" turns stay in documented thinking mode instead of + // sending unsupported effort strings. + supportsDeveloperRole: false, supportsReasoningEffort: true, - reasoningEffortMap: { xhigh: "max" }, - // `tool_choice` returns 400 against DeepSeek when reasoning_effort is set - // (per the issue thread). Tool calls still work without the parameter. + reasoningEffortMap: { minimal: "high", low: "high", medium: "high", high: "high", xhigh: "max" }, + maxTokensField: "max_tokens", + // DeepSeek V4 thinking mode rejects the `tool_choice` control parameter. + // Tool calls still work without it; the API defaults to auto when tools exist. supportsToolChoice: false, + // DeepSeek V4's OpenAI format docs enable thinking with both the toggle and + // reasoning_effort. Keep the toggle explicit for built-in models. + extraBody: { thinking: { type: "enabled" } }, // DeepSeek emits chain-of-thought via `reasoning_content` and requires it // to round-trip on assistant tool-call messages so the model can resume // from prior thinking (interleaved.field=reasoning_content on models.dev, // matches the kimi/openrouter handling already in detectCompat). reasoningContentField: "reasoning_content", requiresReasoningContentForToolCalls: true, + requiresAssistantContentForToolCalls: true, }, }), ]; diff --git a/packages/ai/src/providers/openai-completions-compat.ts b/packages/ai/src/providers/openai-completions-compat.ts index 62d3a6883..3882fc7aa 100644 --- a/packages/ai/src/providers/openai-completions-compat.ts +++ b/packages/ai/src/providers/openai-completions-compat.ts @@ -79,7 +79,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB baseUrl.includes("deepseek.com") || lowerId.includes("deepseek") || lowerName.includes("deepseek"); - + const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com"); + const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning); const isNonStandard = isCerebras || provider === "xai" || @@ -102,7 +103,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB provider === "mistral" || baseUrl.includes("mistral.ai") || baseUrl.includes("chutes.ai") || - baseUrl.includes("fireworks.ai"); + baseUrl.includes("fireworks.ai") || + isDirectDeepseekApi; const isGrok = provider === "xai" || baseUrl.includes("api.x.ai"); const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai"); @@ -162,7 +164,13 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB xhigh: "default", } satisfies Partial>) : isDeepseekFamily && model.reasoning - ? { xhigh: "max" } + ? ({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + } satisfies Partial>) : {}; return { @@ -173,8 +181,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB reasoningEffortMap, supportsUsageInStreaming: !isCerebras, disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel, - disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning), - supportsToolChoice: true, + disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter, + supportsToolChoice: !isDirectDeepseekReasoning, maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens", requiresToolResultName: isMistral, requiresAssistantAfterToolResult: false, @@ -204,11 +212,11 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB // DeepSeek V4 rejects synthetic reasoning_content placeholders (".") on tool-call turns. // Kimi and OpenRouter accept them when actual reasoning is unavailable. allowsSyntheticReasoningContentForToolCalls: !isDeepseekFamily || !model.reasoning, - requiresAssistantContentForToolCalls: isKimiModel, + requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning, openRouterRouting: undefined, vercelGatewayRouting: undefined, supportsStrictMode: detectStrictModeSupport(provider, baseUrl), - extraBody: undefined, + extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined, toolStrictMode: isCerebras ? "all_strict" : "mixed", }; } @@ -235,7 +243,7 @@ export function resolveOpenAICompat( supportsMultipleSystemMessages: model.compat.supportsMultipleSystemMessages ?? detected.supportsMultipleSystemMessages, supportsReasoningEffort: model.compat.supportsReasoningEffort ?? detected.supportsReasoningEffort, - reasoningEffortMap: model.compat.reasoningEffortMap ?? detected.reasoningEffortMap, + reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) }, supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming, supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice, maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField, @@ -259,7 +267,7 @@ export function resolveOpenAICompat( openRouterRouting: model.compat.openRouterRouting ?? detected.openRouterRouting, vercelGatewayRouting: model.compat.vercelGatewayRouting ?? detected.vercelGatewayRouting, supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode, - extraBody: model.compat.extraBody, + extraBody: model.compat.extraBody ?? detected.extraBody, toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode, }; } diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index c4899f9f3..48b8407b5 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -47,7 +47,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { // Fix 1: reasoningEffortMap for DeepSeek-family on any provider // ---------------------------------------------------------------- describe("reasoningEffortMap (Fix 1)", () => { - it("maps xhigh → max for DeepSeek-family on opencode-go", () => { + it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => { const compat = detectCompat( deepseekModel({ provider: "opencode-go", @@ -55,10 +55,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => { id: "deepseek-v4-flash", }), ); - expect(compat.reasoningEffortMap.xhigh).toBe("max"); + expect(compat.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }); }); - it("maps xhigh → max for DeepSeek-family on NVIDIA", () => { + it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => { const compat = detectCompat( deepseekModel({ provider: "nvidia", @@ -66,10 +72,16 @@ describe("DeepSeek reasoning_content tool-call replay", () => { id: "deepseek-ai/deepseek-v4-flash", }), ); - expect(compat.reasoningEffortMap.xhigh).toBe("max"); + expect(compat.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }); }); - it("maps xhigh → max for DeepSeek on the official endpoint", () => { + it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => { const compat = detectCompat( deepseekModel({ provider: "deepseek", @@ -77,7 +89,13 @@ describe("DeepSeek reasoning_content tool-call replay", () => { id: "deepseek-v4-pro", }), ); - expect(compat.reasoningEffortMap.xhigh).toBe("max"); + expect(compat.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }); }); it("does NOT map xhigh for non-DeepSeek models", () => { diff --git a/packages/ai/test/issue-1207-repro.test.ts b/packages/ai/test/issue-1207-repro.test.ts new file mode 100644 index 000000000..85e31868d --- /dev/null +++ b/packages/ai/test/issue-1207-repro.test.ts @@ -0,0 +1,105 @@ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "@oh-my-pi/pi-ai/models"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat"; +import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import * as z from "zod/v4"; + +const echoTool: Tool = { + name: "echo", + description: "Echo input", + parameters: z.object({ text: z.string() }), +}; + +const contextWithTools: Context = { + messages: [{ role: "user", content: "call echo", timestamp: Date.now() }], + tools: [echoTool], +}; + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +async function capturePayload(model: Model<"openai-completions">): Promise> { + const { promise, resolve } = Promise.withResolvers(); + streamOpenAICompletions(model, contextWithTools, { + apiKey: "test-key", + signal: abortedSignal(), + reasoning: "minimal", + toolChoice: "auto", + maxTokens: 123, + onPayload: payload => resolve(payload), + }); + return (await promise) as Record; +} + +function customDeepseekFlash(): Model<"openai-completions"> { + return { + ...getBundledModel("openai", "gpt-4o-mini"), + api: "openai-completions", + id: "deepseek-v4-flash", + name: "DeepSeek V4 Flash", + provider: "ds", + baseUrl: "https://api.deepseek.com/v1", + reasoning: true, + compat: { + supportsReasoningEffort: true, + reasoningEffortMap: { xhigh: "max" }, + }, + }; +} + +describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { + it("detects the documented direct DeepSeek V4 compat shape", () => { + const model = getBundledModel("deepseek", "deepseek-v4-flash") as Model<"openai-completions">; + const compat = detectOpenAICompat(model); + + expect(compat.supportsToolChoice).toBe(false); + expect(compat.maxTokensField).toBe("max_tokens"); + expect(compat.extraBody).toEqual({ thinking: { type: "enabled" } }); + expect(compat.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }); + }); + + it("merges partial user reasoning maps with DeepSeek defaults", () => { + const compat = resolveOpenAICompat(customDeepseekFlash()); + + expect(compat.supportsToolChoice).toBe(false); + expect(compat.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + xhigh: "max", + }); + }); + + it("omits tool_choice but preserves documented reasoning when tools are present", async () => { + const body = await capturePayload(customDeepseekFlash()); + + expect(body.tools).toBeDefined(); + expect(body.tool_choice).toBeUndefined(); + expect(body.reasoning_effort).toBe("high"); + expect(body.thinking).toEqual({ type: "enabled" }); + expect(body.max_tokens).toBe(123); + expect(body.max_completion_tokens).toBeUndefined(); + }); + + it("preserves OpenRouter reasoning when tool_choice auto is present", async () => { + const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">; + const compat = detectOpenAICompat(model); + const body = await capturePayload(model); + + expect(compat.disableReasoningOnToolChoice).toBe(false); + expect(body.tools).toBeDefined(); + expect(body.tool_choice).toBe("auto"); + expect(body.reasoning).toEqual({ effort: "high" }); + expect(body.reasoning_effort).toBeUndefined(); + }); +}); diff --git a/packages/ai/test/issue-830-repro.test.ts b/packages/ai/test/issue-830-repro.test.ts index 9832110db..24d5deaab 100644 --- a/packages/ai/test/issue-830-repro.test.ts +++ b/packages/ai/test/issue-830-repro.test.ts @@ -33,15 +33,25 @@ describe("deepseek built-in provider (issue #830)", () => { expect(descriptor?.modelsDevKey).toBe("deepseek"); expect(descriptor?.api).toBe("openai-completions"); expect(descriptor?.baseUrl).toBe("https://api.deepseek.com"); - // Per-model compat: deepseek-v4 reasoning models leak chat-template tool-call markers - // (#798) and 400 on tool_choice when xhigh effort is used (#830 thread). Reasoning content - // must round-trip on tool calls (interleaved.field=reasoning_content from models.dev). + // Per-model compat: DeepSeek V4 supports thinking-mode tool calls, but only + // with `high`/`max` effort, no explicit `tool_choice`, max_tokens, and + // reasoning_content replay. const compat = descriptor?.api === "openai-completions" ? (descriptor.compat as OpenAICompat | undefined) : undefined; + expect(compat?.supportsDeveloperRole).toBe(false); expect(compat?.supportsReasoningEffort).toBe(true); expect(compat?.supportsToolChoice).toBe(false); + expect(compat?.maxTokensField).toBe("max_tokens"); expect(compat?.requiresReasoningContentForToolCalls).toBe(true); + expect(compat?.requiresAssistantContentForToolCalls).toBe(true); expect(compat?.reasoningContentField).toBe("reasoning_content"); - expect(compat?.reasoningEffortMap?.xhigh).toBe("max"); + expect(compat?.extraBody).toEqual({ thinking: { type: "enabled" } }); + expect(compat?.reasoningEffortMap).toMatchObject({ + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }); }); });