fix: added reasoning effort support for qwen templates

- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates.
- Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers.
- Enabled default reasoning enforcement and updated cache provider invalidation.
- Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
can1357
2026-08-19 00:47:11 +02:00
parent 8500092296
commit bf490ae024
16 changed files with 317 additions and 6 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed thinking effort selections being ignored for local Qwen 3.8+ models on llama.cpp and vLLM: the Qwen chat-completions dialects only toggled `enable_thinking`, so the chat template always reasoned at its `xhigh` default no matter which level was selected. The encoder now routes the requested effort onto the template's `reasoning_effort` kwarg (`chat_template_kwargs` for both Qwen dialects, plus the top-level field newer llama.cpp builds map natively).
## [17.3.7] - 2026-08-17
### Changed
+24 -2
View File
@@ -747,7 +747,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
thinking?: { type: "enabled" | "disabled"; effort?: string; keep?: "all" };
enable_thinking?: boolean;
preserve_thinking?: boolean;
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean; reasoning_effort?: string };
reasoning?: { effort?: string } | { enabled: false };
reasoning_effort?: string | null;
service_tier?: ServiceTier;
@@ -1049,12 +1049,34 @@ export function applyChatCompletionsCompatPolicy(params: OpenAICompletionsParams
break;
case "qwen-enable-thinking-false":
params.enable_thinking = true;
// Qwen 3.8+ templates steer thinking depth via the
// `reasoning_effort` kwarg (low/medium/xhigh, template default
// xhigh) — without it every effort selection lands on xhigh.
// Twin emission mirrors `preserve_thinking` above: newer
// llama.cpp builds map the top-level OpenAI field into the
// template, older builds and Alibaba-style local servers read
// only the kwargs copy. The `qwen-chat-template` dialect (NIM,
// vLLM/SGLang) rides kwargs alone — NIM's request schema
// rejects unknown top-level fields (#2299).
if (policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined) {
params.reasoning_effort = reasoning.wireEffort;
params.chat_template_kwargs = {
...params.chat_template_kwargs,
reasoning_effort: reasoning.wireEffort,
};
}
break;
case "qwen-template-false":
// Spread so the `preserve_thinking` kwarg hoisted above
// survives the merge — a bare `{ enable_thinking: true }`
// would clobber it.
params.chat_template_kwargs = { ...params.chat_template_kwargs, enable_thinking: true };
params.chat_template_kwargs = {
...params.chat_template_kwargs,
enable_thinking: true,
...(policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined
? { reasoning_effort: reasoning.wireEffort }
: {}),
};
break;
case "openrouter-enabled-false":
if (reasoning.wireEffort !== undefined) {
@@ -55,6 +55,7 @@ const compat: ResolvedOpenAICompat = {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
@@ -199,4 +199,70 @@ describe("OpenAI compat policy", () => {
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBeUndefined();
});
function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> {
return buildModel({
id,
name: id,
api: "openai-completions",
provider,
baseUrl,
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 262_144,
maxTokens: 32_768,
} satisfies ModelSpec<"openai-completions">);
}
it("routes local Qwen3.8 effort selections onto the chat template (llama.cpp qwen dialect)", () => {
// Regression: the qwen dialects used to emit only `enable_thinking: true`,
// so every effort selection ran at the template's xhigh default.
const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
);
// Twin emission: top-level for newer llama.cpp builds, kwargs for
// older builds — and the preserve_thinking kwarg must survive.
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBe(effort);
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: effort });
expect(params.preserve_thinking).toBe(true);
}
});
it("routes local Qwen3.8 effort selections via chat_template_kwargs only on vLLM", () => {
// vLLM's renderer reads chat_template_kwargs; NIM-style schemas reject
// unknown top-level fields, so nothing may ride top-level here.
const model = localQwenModel("qwen3.8-27b", "vllm", "http://127.0.0.1:8000/v1");
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }),
);
expect(params.enable_thinking).toBeUndefined();
expect(params.reasoning_effort).toBeUndefined();
expect(params.chat_template_kwargs).toEqual({
preserve_thinking: true,
enable_thinking: true,
reasoning_effort: Effort.Medium,
});
});
it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => {
// Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would
// inject an undefined template variable for zero benefit.
const model = localQwenModel("qwen-3.6-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.High }),
);
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBeUndefined();
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
});
});
@@ -202,6 +202,7 @@ describe("openai-completions compatibility", () => {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
@@ -43,6 +43,7 @@ const compat: ResolvedOpenAICompat = {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},