diff --git a/docs/models.md b/docs/models.md index dd709ed7b..4778d68f3 100644 --- a/docs/models.md +++ b/docs/models.md @@ -517,6 +517,7 @@ Reasoning / thinking: - `supportsReasoningParams` — whether request shaping may send reasoning params at all. Default: auto (off for GitHub Copilot chat-completions). - `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh|max`) to provider-specific strings (e.g. Fireworks GLM maps `minimal -> "none"`). - `thinkingFormat` — request shape for thinking: `"openai"` (`reasoning_effort`), `"openrouter"` (`reasoning: { effort }`), `"zai"` (`thinking: { type: "enabled" }`), `"qwen"` (top-level `enable_thinking`), or `"qwen-chat-template"` (`chat_template_kwargs.enable_thinking`). Default: `"openai"`. +- `qwenTemplateReasoningEffort` — route the selected effort onto the Qwen 3.8+ chat template's `reasoning_effort` kwarg (`chat_template_kwargs.reasoning_effort`, plus the top-level field on the `qwen` dialect). Default: auto (on for Qwen 3.8+ ids on local non-Ollama backends). Set `false` for strict servers that reject unknown `chat_template_kwargs`; effort selections are then not sent for the Qwen dialects and the template runs at its own default. - `reasoningContentField` — assistant field carrying chain-of-thought: `"reasoning_content"`, `"reasoning"`, or `"reasoning_text"`. Default: auto. - `requiresReasoningContentForToolCalls` — assistant tool-call turns must round-trip the reasoning field (DeepSeek-R1, Kimi, OpenRouter when reasoning is on). Default: `false`. - `allowsSyntheticReasoningContentForToolCalls` — allow a placeholder reasoning field when a prior assistant tool-call turn lacks provider reasoning content. Default: `true`; set `false` for providers that validate the exact reasoning value. diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 824bdf312..1192cb0a7 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed local OpenAI-compatible servers with strict `chat_template_kwargs` whitelists (e.g. NInfer) failing every Qwen 3.8+ turn with `400 chat_template_kwargs.reasoning_effort is not supported` after the effort routing fix: the reasoning-effort fallback now recognizes a rejection of the kwargs spelling itself, retries with the kwarg stripped while keeping the effort on the standard top-level `reasoning_effort` field (hoisting it there for the kwargs-only vLLM dialect), and remembers the shape for the rest of the session. Value-level rejections and drops now also update the `chat_template_kwargs.reasoning_effort` twin instead of leaving a stale effort for kwargs-reading renderers, and unknown-parameter 400s naming `reasoning_effort` are recognized as effort rejections. + ## [17.3.8] - 2026-08-19 ### Changed diff --git a/packages/ai/src/providers/openai-reasoning-fallback.ts b/packages/ai/src/providers/openai-reasoning-fallback.ts index bd9a4ed57..b6ccb12d4 100644 --- a/packages/ai/src/providers/openai-reasoning-fallback.ts +++ b/packages/ai/src/providers/openai-reasoning-fallback.ts @@ -1,8 +1,17 @@ import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import type { CapturedHttpErrorResponse } from "../utils/http-inspector"; +/** + * Fallback marker: the server rejected the `chat_template_kwargs.reasoning_effort` + * spelling itself (strict kwargs whitelists — Ninfer-style servers), not the + * effort value. Apply strips the kwarg and hoists the value onto the top-level + * `reasoning_effort` field when that spelling is absent. + * @internal + */ +export const STRIP_TEMPLATE_KWARG_REASONING_EFFORT = Symbol("strip-template-kwarg-reasoning-effort"); + /** @internal */ -export type OpenAIReasoningEffortFallback = string | null; +export type OpenAIReasoningEffortFallback = string | null | typeof STRIP_TEMPLATE_KWARG_REASONING_EFFORT; /** @internal */ export interface OpenAIReasoningEffortFallbackState { @@ -73,7 +82,23 @@ export function readOpenAIReasoningEffort(params: unknown): string | undefined { if (!isRecord(params)) return undefined; if (typeof params.reasoning_effort === "string") return params.reasoning_effort; const reasoning = params.reasoning; - return isRecord(reasoning) && typeof reasoning.effort === "string" ? reasoning.effort : undefined; + if (isRecord(reasoning) && typeof reasoning.effort === "string") return reasoning.effort; + return readTemplateKwargReasoningEffort(params); +} + +function readTemplateKwargReasoningEffort(params: Record): string | undefined { + const kwargs = params.chat_template_kwargs; + return isRecord(kwargs) && typeof kwargs.reasoning_effort === "string" ? kwargs.reasoning_effort : undefined; +} + +/** Remove `chat_template_kwargs.reasoning_effort`, dropping the kwargs object when it becomes empty. */ +function deleteTemplateKwargReasoningEffort(kwargs: Record, parent: Record): void { + delete kwargs.reasoning_effort; + for (const key in kwargs) { + void key; + return; + } + delete parent.chat_template_kwargs; } function deleteReasoningEffort(reasoning: Record, parent: Record): boolean { @@ -89,6 +114,17 @@ function deleteReasoningEffort(reasoning: Record, parent: Recor /** @internal */ export function applyOpenAIReasoningEffortFallback(params: unknown, fallback: OpenAIReasoningEffortFallback): boolean { if (!isRecord(params)) return false; + if (fallback === STRIP_TEMPLATE_KWARG_REASONING_EFFORT) { + const kwargs = params.chat_template_kwargs; + if (!isRecord(kwargs) || typeof kwargs.reasoning_effort !== "string") return false; + const effort = kwargs.reasoning_effort; + deleteTemplateKwargReasoningEffort(kwargs, params); + // The `qwen-chat-template` dialect rides kwargs alone; keep the effort + // selection alive on the standard OpenAI field (Ninfer-style servers + // accept it, vLLM-style renderers ignore it). + if (typeof params.reasoning_effort !== "string") params.reasoning_effort = effort; + return true; + } let changed = false; if (typeof params.reasoning_effort === "string") { if (fallback === null) { @@ -107,6 +143,17 @@ export function applyOpenAIReasoningEffortFallback(params: unknown, fallback: Op changed = true; } } + // Keep the Qwen template kwarg twin in lockstep — a value remap or drop + // must not leave a stale effort for kwargs-reading renderers. + const kwargs = params.chat_template_kwargs; + if (isRecord(kwargs) && typeof kwargs.reasoning_effort === "string") { + if (fallback === null) { + deleteTemplateKwargReasoningEffort(kwargs, params); + } else { + kwargs.reasoning_effort = fallback; + } + changed = true; + } return changed; } @@ -165,13 +212,17 @@ function isInvalidReasoningEffortError( if (/reasoning[_ ]content/i.test(message) && !REASONING_EFFORT_FIELD_PATTERN.test(message)) return false; if (/invalid[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) return true; if ( - /(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|must be|expected)/i.test( + /(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test( message, ) ) { return true; } - if (/(?:unsupported|not supported)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) { + if ( + /(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test( + message, + ) + ) { return true; } // Gateways put the rejected value first (`level "none" not supported`), the @@ -258,6 +309,35 @@ function nearestEnabledReasoningFallback(currentEffort: string, allowed: Set { expect(params.reasoning_effort).toBeUndefined(); }); - function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> { + function localQwenModel( + id: string, + provider: string, + baseUrl: string, + compat?: OpenAICompat, + ): Model<"openai-completions"> { return buildModel({ id, name: id, @@ -212,6 +217,7 @@ describe("OpenAI compat policy", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 262_144, maxTokens: 32_768, + compat, } satisfies ModelSpec<"openai-completions">); } @@ -252,6 +258,24 @@ describe("OpenAI compat policy", () => { }); }); + it("honors a user compat override disabling the template effort dialect", () => { + // Escape hatch for strict local servers (Ninfer-style) that reject + // unknown chat_template_kwargs: `qwenTemplateReasoningEffort: false` in + // models.yml must suppress the kwarg and revert to the pre-effort wire + // shape without disturbing thinking or preserve_thinking. + const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1", { + qwenTemplateReasoningEffort: false, + }); + const params = chatParams(); + applyChatCompletionsCompatPolicy( + params, + resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }), + ); + expect(params.enable_thinking).toBe(true); + expect(params.reasoning_effort).toBeUndefined(); + expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true }); + }); + it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => { // Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would // inject an undefined template variable for zero benefit. diff --git a/packages/ai/test/openai-reasoning-effort-fallback.test.ts b/packages/ai/test/openai-reasoning-effort-fallback.test.ts index 07998ba2e..b0c9ceb14 100644 --- a/packages/ai/test/openai-reasoning-effort-fallback.test.ts +++ b/packages/ai/test/openai-reasoning-effort-fallback.test.ts @@ -133,6 +133,53 @@ function summaryReasoningErrorResponse(): Response { ); } +/** + * Ninfer-style strict kwargs whitelist: the server rejects the + * `chat_template_kwargs.reasoning_effort` spelling itself, not the value. + */ +function templateKwargRejectionResponse(): Response { + return new Response( + JSON.stringify({ + error: { + message: "chat_template_kwargs.reasoning_effort is not supported", + type: "invalid_request_error", + code: "unknown_parameter", + param: "chat_template_kwargs.reasoning_effort", + }, + }), + { status: 400, headers: { "content-type": "application/json" } }, + ); +} + +function templateKwargValueRejectionResponse(): Response { + return new Response( + JSON.stringify({ + error: { + message: "chat_template_kwargs.reasoning_effort: 'xhigh' is not supported, valid levels: low, medium, high", + type: "invalid_request_error", + param: "chat_template_kwargs.reasoning_effort", + }, + }), + { status: 400, headers: { "content-type": "application/json" } }, + ); +} + +/** Local Qwen 3.8 model whose auto-compat routes effort onto the template kwarg. */ +function createLocalQwenModel(provider: string, baseUrl: string): Model<"openai-completions"> { + return buildModel({ + id: "qwen3.8-27b", + name: "Qwen3.8 27B (local)", + api: "openai-completions", + provider, + baseUrl, + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262_144, + maxTokens: 32_768, + }); +} + function parseJsonBody(init: RequestInit | undefined): Record { return JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; } @@ -398,4 +445,103 @@ describe("OpenAI reasoning effort fallback retry", () => { expect(result.errorStatus).toBe(400); expect(attempts).toBe(1); }); + + it("strips a rejected chat_template_kwargs.reasoning_effort, keeps the top-level twin, and remembers it", async () => { + const bodies: Record[] = []; + const fetchMock: FetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit): Promise => { + const body = parseJsonBody(init); + bodies.push(body); + return bodies.length === 1 ? templateKwargRejectionResponse() : createChatSseResponse(); + }, + { preconnect: fetch.preconnect }, + ); + const providerSessionState = new Map(); + const model = createLocalQwenModel("llama.cpp", "http://127.0.0.1:8080/v1"); + + const first = await streamOpenAICompletions(model, testContext, { + apiKey: "test-key", + fetch: fetchMock, + reasoning: "medium", + providerSessionState, + }).result(); + + expect(first.stopReason).toBe("stop"); + expect(bodies).toHaveLength(2); + // The qwen dialect twin-emits; the rejected kwarg must vanish while the + // top-level field keeps the user's effort selection alive. + expect(bodies[0]!.reasoning_effort).toBe("medium"); + expect(bodies[0]!.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: "medium" }); + expect(bodies[1]!.reasoning_effort).toBe("medium"); + expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true }); + + // Remembered per session: the next request pre-strips without a 400. + const second = await streamOpenAICompletions(model, testContext, { + apiKey: "test-key", + fetch: fetchMock, + reasoning: "medium", + providerSessionState, + }).result(); + expect(second.stopReason).toBe("stop"); + expect(bodies).toHaveLength(3); + expect(bodies[2]!.reasoning_effort).toBe("medium"); + expect(bodies[2]!.chat_template_kwargs).toEqual({ preserve_thinking: true }); + }); + + it("hoists the effort onto the top-level field when the kwargs-only dialect is rejected", async () => { + const bodies: Record[] = []; + const fetchMock: FetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit): Promise => { + const body = parseJsonBody(init); + bodies.push(body); + return bodies.length === 1 ? templateKwargRejectionResponse() : createChatSseResponse(); + }, + { preconnect: fetch.preconnect }, + ); + const model = createLocalQwenModel("vllm", "http://127.0.0.1:8000/v1"); + + const result = await streamOpenAICompletions(model, testContext, { + apiKey: "test-key", + fetch: fetchMock, + reasoning: "medium", + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(bodies).toHaveLength(2); + // vLLM dialect rides kwargs alone — nothing top-level on the first try. + expect(bodies[0]!.reasoning_effort).toBeUndefined(); + expect(bodies[0]!.chat_template_kwargs).toEqual({ + preserve_thinking: true, + enable_thinking: true, + reasoning_effort: "medium", + }); + expect(bodies[1]!.reasoning_effort).toBe("medium"); + expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true, enable_thinking: true }); + }); + + it("remaps a rejected kwargs effort value in both spellings when the error lists allowed levels", async () => { + const bodies: Record[] = []; + const fetchMock: FetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit): Promise => { + const body = parseJsonBody(init); + bodies.push(body); + return bodies.length === 1 ? templateKwargValueRejectionResponse() : createChatSseResponse(); + }, + { preconnect: fetch.preconnect }, + ); + const model = createLocalQwenModel("llama.cpp", "http://127.0.0.1:8080/v1"); + + const result = await streamOpenAICompletions(model, testContext, { + apiKey: "test-key", + fetch: fetchMock, + reasoning: "xhigh", + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(bodies).toHaveLength(2); + expect(bodies[0]!.reasoning_effort).toBe("xhigh"); + expect(bodies[1]!.reasoning_effort).toBe("high"); + // The kwargs twin must not keep the stale rejected value. + expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: "high" }); + }); }); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index cd31a58e6..a0e76a2ff 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`. + ### Changed - `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so. diff --git a/packages/coding-agent/src/config/models-config-schema-bundle.ts b/packages/coding-agent/src/config/models-config-schema-bundle.ts index 458eeaca1..89f4f4806 100644 --- a/packages/coding-agent/src/config/models-config-schema-bundle.ts +++ b/packages/coding-agent/src/config/models-config-schema-bundle.ts @@ -42,6 +42,7 @@ export const getModelsConfigSchemaBundle = once(() => { "disableReasoningOnForcedToolChoice?": "boolean", "disableReasoningOnToolChoice?": "boolean", "thinkingFormat?": '"openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template"', + "qwenTemplateReasoningEffort?": "boolean", "openRouterRouting?": OpenRouterRoutingSchema, "vercelGatewayRouting?": VercelGatewayRoutingSchema, "extraBody?": { "[string]": "unknown" },