fix(ai): dropped nanogpt :tools route on deepseek
The :tools model suffix engages NanoGPT's server-side tool-call parser, which 502s with code malformed_tool_call on complex DeepSeek payloads (observed reliably on todo_write). The default route forwards delta.content (including DSML envelope leaks) which our StreamMarkupHealing already heals into a structured tool call. The DSML allowlist still covers nanogpt and the parallel-index fix remains; only the :tools suffix is removed. Fixes #1488
This commit is contained in:
@@ -10,6 +10,7 @@
|
||||
- Fixed GLM-5.x coding-plan OpenAI-compatible streams to use a longer default watchdog window, avoiding spurious `OpenAI completions stream stalled while waiting for the next event` errors during slow `glm-5.1` thinking/output phases. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494))
|
||||
- Fixed `zhipu-coding-plan` model discovery and credential validation to use the dedicated GLM Coding Plan endpoint (`https://open.bigmodel.cn/api/coding/paas/v4`) instead of the general BigModel endpoint, preventing requests from consuming ordinary account balance. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494))
|
||||
- Fixed DeepSeek tool calls failing on NanoGPT (e.g. `nanogpt/deepseek/deepseek-v4-pro` with reasoning enabled) by routing tool-bearing DeepSeek requests through NanoGPT's `:tools` model route and adding `nanogpt` to the DSML leak allowlist so streamed `<|DSML|tool_calls>...</|DSML|tool_calls>` envelopes are healed into structured tool calls instead of being passed through as visible text. ([#1488](https://github.com/can1357/oh-my-pi/issues/1488))
|
||||
- Fixed DeepSeek tool calls failing on NanoGPT (e.g. `nanogpt/deepseek/deepseek-v4-pro` with reasoning enabled) by adding `nanogpt` to the DSML leak allowlist so streamed `<|DSML|tool_calls>...</|DSML|tool_calls>` envelopes are healed into structured tool calls instead of being passed through as visible text. The `:tools` model suffix is no longer appended on NanoGPT; that route triggered NanoGPT's server-side tool-call parser and 502'd with `code: "malformed_tool_call"` on complex tool schemas (`todo_write`) — the default route forwards `delta.content` (including DSML envelopes) which is healed client-side. ([#1488](https://github.com/can1357/oh-my-pi/issues/1488))
|
||||
- Fixed OpenAI-compatible streamed parallel tool calls losing indexed argument deltas by tracking active tool-call blocks by the provider's `tool_calls[].index`; this keeps parallel NanoGPT `read` calls from merging or dropping their `path` arguments. ([#1488](https://github.com/can1357/oh-my-pi/issues/1488))
|
||||
|
||||
## [15.5.11] - 2026-05-29
|
||||
|
||||
@@ -92,24 +92,20 @@ function normalizeMistralToolId(id: string, isMistral: boolean): string {
|
||||
}
|
||||
return normalized;
|
||||
}
|
||||
|
||||
// NanoGPT's default DeepSeek route can attempt server-side tool-call repair and
|
||||
// fail before streaming. `:tools` selects its documented tools-capable route.
|
||||
function shouldUseNanoGptToolsRoute(model: Model<"openai-completions">, context: Context): boolean {
|
||||
return (
|
||||
model.provider === "nanogpt" && !!context.tools?.length && /deepseek/i.test(model.id) && !model.id.includes(":")
|
||||
);
|
||||
}
|
||||
|
||||
// Direct DeepSeek model ids on NanoGPT are routed via the default tools-capable
|
||||
// path. We deliberately do NOT append `:tools` here: with `:tools`, NanoGPT
|
||||
// performs server-side tool-call parsing on the upstream DeepSeek stream and
|
||||
// 502s with `code: "malformed_tool_call"` on more complex tool schemas (issue
|
||||
// #1488). The default route forwards `delta.content` (including any DSML
|
||||
// envelope leaks) which `StreamMarkupHealing` heals into a structured call
|
||||
// client-side.
|
||||
function resolveOpenAICompletionsModelId(
|
||||
model: Model<"openai-completions">,
|
||||
context: Context,
|
||||
options: OpenAICompletionsOptions | undefined,
|
||||
): string {
|
||||
if (model.provider === "firepass") return toFirepassWireModelId(model.id);
|
||||
if (model.provider === "fireworks") return toFireworksWireModelId(model.id);
|
||||
if (model.provider === "openrouter") return applyOpenRouterRoutingVariant(model.id, options?.openrouterVariant);
|
||||
if (shouldUseNanoGptToolsRoute(model, context)) return `${model.id}:tools`;
|
||||
return model.id;
|
||||
}
|
||||
|
||||
@@ -1140,7 +1136,7 @@ function buildParams(
|
||||
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
|
||||
const effectiveMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined);
|
||||
|
||||
const requestModelId = resolveOpenAICompletionsModelId(model, context, options);
|
||||
const requestModelId = resolveOpenAICompletionsModelId(model, options);
|
||||
const params: OpenAICompletionsParams = {
|
||||
model: requestModelId,
|
||||
messages,
|
||||
|
||||
@@ -659,7 +659,10 @@ describe("OpenAI completions provider DSML envelope healing", () => {
|
||||
},
|
||||
).result();
|
||||
|
||||
expect(payload?.model).toBe("deepseek/deepseek-v4-pro:tools");
|
||||
// Issue #1488: `:tools` triggers NanoGPT's server-side tool-call parser
|
||||
// which 502s on complex DeepSeek payloads. We route via the default
|
||||
// path and rely on DSML healing instead.
|
||||
expect(payload?.model).toBe("deepseek/deepseek-v4-pro");
|
||||
expect(payload?.reasoning_effort).toBe("high");
|
||||
expect(payload?.tools).toBeDefined();
|
||||
const text = result.content
|
||||
|
||||
Reference in New Issue
Block a user