fix: added reasoning effort support for qwen templates
- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates. - Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers. - Enabled default reasoning enforcement and updated cache provider invalidation. - Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed thinking effort selections being ignored for local Qwen 3.8+ models on llama.cpp and vLLM: the Qwen chat-completions dialects only toggled `enable_thinking`, so the chat template always reasoned at its `xhigh` default no matter which level was selected. The encoder now routes the requested effort onto the template's `reasoning_effort` kwarg (`chat_template_kwargs` for both Qwen dialects, plus the top-level field newer llama.cpp builds map natively).
|
||||
|
||||
## [17.3.7] - 2026-08-17
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -747,7 +747,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
||||
thinking?: { type: "enabled" | "disabled"; effort?: string; keep?: "all" };
|
||||
enable_thinking?: boolean;
|
||||
preserve_thinking?: boolean;
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean; reasoning_effort?: string };
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
reasoning_effort?: string | null;
|
||||
service_tier?: ServiceTier;
|
||||
@@ -1049,12 +1049,34 @@ export function applyChatCompletionsCompatPolicy(params: OpenAICompletionsParams
|
||||
break;
|
||||
case "qwen-enable-thinking-false":
|
||||
params.enable_thinking = true;
|
||||
// Qwen 3.8+ templates steer thinking depth via the
|
||||
// `reasoning_effort` kwarg (low/medium/xhigh, template default
|
||||
// xhigh) — without it every effort selection lands on xhigh.
|
||||
// Twin emission mirrors `preserve_thinking` above: newer
|
||||
// llama.cpp builds map the top-level OpenAI field into the
|
||||
// template, older builds and Alibaba-style local servers read
|
||||
// only the kwargs copy. The `qwen-chat-template` dialect (NIM,
|
||||
// vLLM/SGLang) rides kwargs alone — NIM's request schema
|
||||
// rejects unknown top-level fields (#2299).
|
||||
if (policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined) {
|
||||
params.reasoning_effort = reasoning.wireEffort;
|
||||
params.chat_template_kwargs = {
|
||||
...params.chat_template_kwargs,
|
||||
reasoning_effort: reasoning.wireEffort,
|
||||
};
|
||||
}
|
||||
break;
|
||||
case "qwen-template-false":
|
||||
// Spread so the `preserve_thinking` kwarg hoisted above
|
||||
// survives the merge — a bare `{ enable_thinking: true }`
|
||||
// would clobber it.
|
||||
params.chat_template_kwargs = { ...params.chat_template_kwargs, enable_thinking: true };
|
||||
params.chat_template_kwargs = {
|
||||
...params.chat_template_kwargs,
|
||||
enable_thinking: true,
|
||||
...(policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined
|
||||
? { reasoning_effort: reasoning.wireEffort }
|
||||
: {}),
|
||||
};
|
||||
break;
|
||||
case "openrouter-enabled-false":
|
||||
if (reasoning.wireEffort !== undefined) {
|
||||
|
||||
@@ -55,6 +55,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -199,4 +199,70 @@ describe("OpenAI compat policy", () => {
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id,
|
||||
name: id,
|
||||
api: "openai-completions",
|
||||
provider,
|
||||
baseUrl,
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 262_144,
|
||||
maxTokens: 32_768,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
it("routes local Qwen3.8 effort selections onto the chat template (llama.cpp qwen dialect)", () => {
|
||||
// Regression: the qwen dialects used to emit only `enable_thinking: true`,
|
||||
// so every effort selection ran at the template's xhigh default.
|
||||
const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
|
||||
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
|
||||
);
|
||||
// Twin emission: top-level for newer llama.cpp builds, kwargs for
|
||||
// older builds — and the preserve_thinking kwarg must survive.
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBe(effort);
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: effort });
|
||||
expect(params.preserve_thinking).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("routes local Qwen3.8 effort selections via chat_template_kwargs only on vLLM", () => {
|
||||
// vLLM's renderer reads chat_template_kwargs; NIM-style schemas reject
|
||||
// unknown top-level fields, so nothing may ride top-level here.
|
||||
const model = localQwenModel("qwen3.8-27b", "vllm", "http://127.0.0.1:8000/v1");
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }),
|
||||
);
|
||||
expect(params.enable_thinking).toBeUndefined();
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
expect(params.chat_template_kwargs).toEqual({
|
||||
preserve_thinking: true,
|
||||
enable_thinking: true,
|
||||
reasoning_effort: Effort.Medium,
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => {
|
||||
// Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would
|
||||
// inject an undefined template variable for zero benefit.
|
||||
const model = localQwenModel("qwen-3.6-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.High }),
|
||||
);
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -202,6 +202,7 @@ describe("openai-completions compatibility", () => {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -43,6 +43,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
Reference in New Issue
Block a user