fix: added reasoning effort support for qwen templates
- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates. - Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers. - Enabled default reasoning enforcement and updated cache provider invalidation. - Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
@@ -14,6 +14,7 @@ import {
|
||||
isMinimaxM3FamilyModelId,
|
||||
isOpenAIGptOssModelId,
|
||||
isOpenAIModelId,
|
||||
isQwen38PlusTemplateEffortModelId,
|
||||
isReasoningGlmModelId,
|
||||
modelFamilyToken,
|
||||
parseAnthropicModel,
|
||||
@@ -30,6 +31,27 @@ describe("isKimiModelId", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("isQwen38PlusTemplateEffortModelId", () => {
|
||||
test("matches Qwen 3.8+ open-weight ids across id shapes and versions", () => {
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-2.4t-a95b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen/qwen3.8-27b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("Qwen3.8-27B-UD-Q6_K_XL")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b:thinking")).toBe(true);
|
||||
// Component-wise version compare: 3.10 sorts after 3.8.
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.10-27b")).toBe(true);
|
||||
});
|
||||
test("rejects pre-3.8 versions, parameter-count lookalikes, and API-only Max SKUs", () => {
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3-8b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen-3.6-27b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.7-plus")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen2.5-coder-7b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen-3.8b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max-preview")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isKimiK26ModelId", () => {
|
||||
test("matches Kimi K2.6 without accepting adjacent versions", () => {
|
||||
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
|
||||
|
||||
@@ -984,3 +984,87 @@ describe("model thinking runtime helpers", () => {
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("Qwen 3.8 local template effort ladder", () => {
|
||||
it("derives the low/medium/xhigh ladder with mandatory effort on local llama.cpp-style backends", () => {
|
||||
const llamaCpp = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "llama.cpp",
|
||||
baseUrl: "http://127.0.0.1:8080/v1",
|
||||
});
|
||||
// Official 3.8 template: reasoning_effort accepts exactly low/medium/xhigh
|
||||
// and raises on `enable_thinking: false` — off must clamp, never disable.
|
||||
expect(llamaCpp.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
});
|
||||
expect(llamaCpp.compat.qwenTemplateReasoningEffort).toBe(true);
|
||||
// Unsupported tiers clamp onto real wire tiers: high floors to medium
|
||||
// (xhigh is a deliberate opt-in), minimal floors to low.
|
||||
expect(clampThinkingLevelForModel(llamaCpp, Effort.High)).toBe(Effort.Medium);
|
||||
expect(clampThinkingLevelForModel(llamaCpp, Effort.Minimal)).toBe(Effort.Low);
|
||||
expect(minimumSupportedEffort(llamaCpp)).toBe(Effort.Low);
|
||||
});
|
||||
|
||||
it("normalizes a stale cached generic ladder to the template ladder", () => {
|
||||
const cached = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
baseUrl: "http://127.0.0.1:8000/v1",
|
||||
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
});
|
||||
expect(cached.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("routes vLLM Qwen through the chat_template_kwargs dialect", () => {
|
||||
// vLLM ignores top-level `enable_thinking`; only chat_template_kwargs
|
||||
// reach the template renderer.
|
||||
const vllm = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
baseUrl: "http://127.0.0.1:8000/v1",
|
||||
});
|
||||
expect(vllm.compat.thinkingFormat).toBe("qwen-chat-template");
|
||||
expect(vllm.compat.reasoningDisableMode).toBe("qwen-template-false");
|
||||
expect(vllm.compat.qwenTemplateReasoningEffort).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps hosted, pre-3.8, and local-Ollama Qwen off the template ladder", () => {
|
||||
const hosted = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "nanogpt",
|
||||
baseUrl: "https://nano-gpt.com/api/v1",
|
||||
});
|
||||
expect(hosted.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(hosted.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
|
||||
|
||||
const qwen36 = createModel({
|
||||
id: "qwen-3.6-27b",
|
||||
api: "openai-completions",
|
||||
provider: "llama.cpp",
|
||||
baseUrl: "http://localhost:8080/v1",
|
||||
});
|
||||
expect(qwen36.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(qwen36.thinking?.requiresEffort).toBeUndefined();
|
||||
|
||||
// Local Ollama renders its own (Go) templates and keeps the native
|
||||
// low/medium/high/max effort vocabulary.
|
||||
const ollama = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "ollama",
|
||||
baseUrl: "http://127.0.0.1:11434/v1",
|
||||
});
|
||||
expect(ollama.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(ollama.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { vllmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
describe("vLLM provider discovery", () => {
|
||||
test("lights up the reasoning dial for Qwen 3.8+ despite silent /v1/models metadata", async () => {
|
||||
// vLLM's /v1/models never advertises reasoning; without the id-based
|
||||
// upgrade a served Qwen3.8 loses its effort dial entirely and always
|
||||
// thinks at the template's xhigh default.
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{ id: "qwen3.8-27b", object: "model", max_model_len: 262144 },
|
||||
{ id: "qwen2.5-coder-7b", object: "model", max_model_len: 131072 },
|
||||
],
|
||||
}),
|
||||
{ status: 200, headers: { "content-type": "application/json" } },
|
||||
);
|
||||
|
||||
const options = vllmModelManagerOptions({ fetch: fetchMock });
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(models?.find(model => model.id === "qwen3.8-27b")).toMatchObject({
|
||||
provider: "vllm",
|
||||
api: "openai-completions",
|
||||
reasoning: true,
|
||||
contextWindow: 262144,
|
||||
});
|
||||
// Non-thinking Qwen generations keep the wire-reported default.
|
||||
expect(models?.find(model => model.id === "qwen2.5-coder-7b")?.reasoning).toBe(false);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user