fix: added reasoning effort support for qwen templates

- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates.
- Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers.
- Enabled default reasoning enforcement and updated cache provider invalidation.
- Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
can1357
2026-08-19 00:47:11 +02:00
parent 8500092296
commit bf490ae024
16 changed files with 317 additions and 6 deletions
@@ -14,6 +14,7 @@ import {
isMinimaxM3FamilyModelId,
isOpenAIGptOssModelId,
isOpenAIModelId,
isQwen38PlusTemplateEffortModelId,
isReasoningGlmModelId,
modelFamilyToken,
parseAnthropicModel,
@@ -30,6 +31,27 @@ describe("isKimiModelId", () => {
});
});
describe("isQwen38PlusTemplateEffortModelId", () => {
test("matches Qwen 3.8+ open-weight ids across id shapes and versions", () => {
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-2.4t-a95b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen/qwen3.8-27b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("Qwen3.8-27B-UD-Q6_K_XL")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b:thinking")).toBe(true);
// Component-wise version compare: 3.10 sorts after 3.8.
expect(isQwen38PlusTemplateEffortModelId("qwen3.10-27b")).toBe(true);
});
test("rejects pre-3.8 versions, parameter-count lookalikes, and API-only Max SKUs", () => {
expect(isQwen38PlusTemplateEffortModelId("qwen3-8b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen-3.6-27b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.7-plus")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen2.5-coder-7b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen-3.8b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max-preview")).toBe(false);
});
});
describe("isKimiK26ModelId", () => {
test("matches Kimi K2.6 without accepting adjacent versions", () => {
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
@@ -984,3 +984,87 @@ describe("model thinking runtime helpers", () => {
});
});
});
describe("Qwen 3.8 local template effort ladder", () => {
it("derives the low/medium/xhigh ladder with mandatory effort on local llama.cpp-style backends", () => {
const llamaCpp = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "llama.cpp",
baseUrl: "http://127.0.0.1:8080/v1",
});
// Official 3.8 template: reasoning_effort accepts exactly low/medium/xhigh
// and raises on `enable_thinking: false` — off must clamp, never disable.
expect(llamaCpp.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
requiresEffort: true,
});
expect(llamaCpp.compat.qwenTemplateReasoningEffort).toBe(true);
// Unsupported tiers clamp onto real wire tiers: high floors to medium
// (xhigh is a deliberate opt-in), minimal floors to low.
expect(clampThinkingLevelForModel(llamaCpp, Effort.High)).toBe(Effort.Medium);
expect(clampThinkingLevelForModel(llamaCpp, Effort.Minimal)).toBe(Effort.Low);
expect(minimumSupportedEffort(llamaCpp)).toBe(Effort.Low);
});
it("normalizes a stale cached generic ladder to the template ladder", () => {
const cached = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "vllm",
baseUrl: "http://127.0.0.1:8000/v1",
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
});
expect(cached.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
requiresEffort: true,
});
});
it("routes vLLM Qwen through the chat_template_kwargs dialect", () => {
// vLLM ignores top-level `enable_thinking`; only chat_template_kwargs
// reach the template renderer.
const vllm = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "vllm",
baseUrl: "http://127.0.0.1:8000/v1",
});
expect(vllm.compat.thinkingFormat).toBe("qwen-chat-template");
expect(vllm.compat.reasoningDisableMode).toBe("qwen-template-false");
expect(vllm.compat.qwenTemplateReasoningEffort).toBe(true);
});
it("keeps hosted, pre-3.8, and local-Ollama Qwen off the template ladder", () => {
const hosted = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "nanogpt",
baseUrl: "https://nano-gpt.com/api/v1",
});
expect(hosted.compat.qwenTemplateReasoningEffort).toBe(false);
expect(hosted.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
const qwen36 = createModel({
id: "qwen-3.6-27b",
api: "openai-completions",
provider: "llama.cpp",
baseUrl: "http://localhost:8080/v1",
});
expect(qwen36.compat.qwenTemplateReasoningEffort).toBe(false);
expect(qwen36.thinking?.requiresEffort).toBeUndefined();
// Local Ollama renders its own (Go) templates and keeps the native
// low/medium/high/max effort vocabulary.
const ollama = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "ollama",
baseUrl: "http://127.0.0.1:11434/v1",
});
expect(ollama.compat.qwenTemplateReasoningEffort).toBe(false);
expect(ollama.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]);
});
});
@@ -0,0 +1,33 @@
import { describe, expect, test } from "bun:test";
import { vllmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
describe("vLLM provider discovery", () => {
test("lights up the reasoning dial for Qwen 3.8+ despite silent /v1/models metadata", async () => {
// vLLM's /v1/models never advertises reasoning; without the id-based
// upgrade a served Qwen3.8 loses its effort dial entirely and always
// thinks at the template's xhigh default.
const fetchMock: FetchImpl = async () =>
new Response(
JSON.stringify({
data: [
{ id: "qwen3.8-27b", object: "model", max_model_len: 262144 },
{ id: "qwen2.5-coder-7b", object: "model", max_model_len: 131072 },
],
}),
{ status: 200, headers: { "content-type": "application/json" } },
);
const options = vllmModelManagerOptions({ fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
expect(models?.find(model => model.id === "qwen3.8-27b")).toMatchObject({
provider: "vllm",
api: "openai-completions",
reasoning: true,
contextWindow: 262144,
});
// Non-thinking Qwen generations keep the wire-reported default.
expect(models?.find(model => model.id === "qwen2.5-coder-7b")?.reasoning).toBe(false);
});
});