fix: added reasoning effort support for qwen templates
- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates. - Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers. - Enabled default reasoning enforcement and updated cache provider invalidation. - Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed thinking effort selections being ignored for local Qwen 3.8+ models on llama.cpp and vLLM: the Qwen chat-completions dialects only toggled `enable_thinking`, so the chat template always reasoned at its `xhigh` default no matter which level was selected. The encoder now routes the requested effort onto the template's `reasoning_effort` kwarg (`chat_template_kwargs` for both Qwen dialects, plus the top-level field newer llama.cpp builds map natively).
|
||||
|
||||
## [17.3.7] - 2026-08-17
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -747,7 +747,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
||||
thinking?: { type: "enabled" | "disabled"; effort?: string; keep?: "all" };
|
||||
enable_thinking?: boolean;
|
||||
preserve_thinking?: boolean;
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean; reasoning_effort?: string };
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
reasoning_effort?: string | null;
|
||||
service_tier?: ServiceTier;
|
||||
@@ -1049,12 +1049,34 @@ export function applyChatCompletionsCompatPolicy(params: OpenAICompletionsParams
|
||||
break;
|
||||
case "qwen-enable-thinking-false":
|
||||
params.enable_thinking = true;
|
||||
// Qwen 3.8+ templates steer thinking depth via the
|
||||
// `reasoning_effort` kwarg (low/medium/xhigh, template default
|
||||
// xhigh) — without it every effort selection lands on xhigh.
|
||||
// Twin emission mirrors `preserve_thinking` above: newer
|
||||
// llama.cpp builds map the top-level OpenAI field into the
|
||||
// template, older builds and Alibaba-style local servers read
|
||||
// only the kwargs copy. The `qwen-chat-template` dialect (NIM,
|
||||
// vLLM/SGLang) rides kwargs alone — NIM's request schema
|
||||
// rejects unknown top-level fields (#2299).
|
||||
if (policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined) {
|
||||
params.reasoning_effort = reasoning.wireEffort;
|
||||
params.chat_template_kwargs = {
|
||||
...params.chat_template_kwargs,
|
||||
reasoning_effort: reasoning.wireEffort,
|
||||
};
|
||||
}
|
||||
break;
|
||||
case "qwen-template-false":
|
||||
// Spread so the `preserve_thinking` kwarg hoisted above
|
||||
// survives the merge — a bare `{ enable_thinking: true }`
|
||||
// would clobber it.
|
||||
params.chat_template_kwargs = { ...params.chat_template_kwargs, enable_thinking: true };
|
||||
params.chat_template_kwargs = {
|
||||
...params.chat_template_kwargs,
|
||||
enable_thinking: true,
|
||||
...(policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined
|
||||
? { reasoning_effort: reasoning.wireEffort }
|
||||
: {}),
|
||||
};
|
||||
break;
|
||||
case "openrouter-enabled-false":
|
||||
if (reasoning.wireEffort !== undefined) {
|
||||
|
||||
@@ -55,6 +55,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -199,4 +199,70 @@ describe("OpenAI compat policy", () => {
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id,
|
||||
name: id,
|
||||
api: "openai-completions",
|
||||
provider,
|
||||
baseUrl,
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 262_144,
|
||||
maxTokens: 32_768,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
it("routes local Qwen3.8 effort selections onto the chat template (llama.cpp qwen dialect)", () => {
|
||||
// Regression: the qwen dialects used to emit only `enable_thinking: true`,
|
||||
// so every effort selection ran at the template's xhigh default.
|
||||
const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
|
||||
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
|
||||
);
|
||||
// Twin emission: top-level for newer llama.cpp builds, kwargs for
|
||||
// older builds — and the preserve_thinking kwarg must survive.
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBe(effort);
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: effort });
|
||||
expect(params.preserve_thinking).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("routes local Qwen3.8 effort selections via chat_template_kwargs only on vLLM", () => {
|
||||
// vLLM's renderer reads chat_template_kwargs; NIM-style schemas reject
|
||||
// unknown top-level fields, so nothing may ride top-level here.
|
||||
const model = localQwenModel("qwen3.8-27b", "vllm", "http://127.0.0.1:8000/v1");
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }),
|
||||
);
|
||||
expect(params.enable_thinking).toBeUndefined();
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
expect(params.chat_template_kwargs).toEqual({
|
||||
preserve_thinking: true,
|
||||
enable_thinking: true,
|
||||
reasoning_effort: Effort.Medium,
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => {
|
||||
// Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would
|
||||
// inject an undefined template variable for zero benefit.
|
||||
const model = localQwenModel("qwen-3.6-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.High }),
|
||||
);
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.reasoning_effort).toBeUndefined();
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -202,6 +202,7 @@ describe("openai-completions compatibility", () => {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -43,6 +43,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning.
|
||||
|
||||
## [17.3.6] - 2026-08-17
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
isKimiModelId,
|
||||
isMimoModelIdOrName,
|
||||
isOpenAISamplingRestrictedModelId,
|
||||
isQwen38PlusTemplateEffortModelId,
|
||||
isQwenModelId,
|
||||
} from "../identity/family";
|
||||
import type {
|
||||
@@ -464,7 +465,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
? "zai"
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
: isQwen && isNvidiaNim
|
||||
: isQwen && (isNvidiaNim || provider === "vllm")
|
||||
? "qwen-chat-template"
|
||||
: isQwen && isFireworks
|
||||
? "openai"
|
||||
@@ -587,6 +588,18 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
// parameter, so the flag stays a no-op outside the Qwen path.
|
||||
qwenPreserveThinking:
|
||||
(thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && isLocalOpenAICompatBackend,
|
||||
// Qwen 3.8+ templates steer thinking depth via the `reasoning_effort`
|
||||
// template kwarg (low/medium/xhigh, default xhigh); without routing the
|
||||
// requested effort there, the enable_thinking toggle alone leaves the
|
||||
// model at xhigh no matter what the user selects.
|
||||
// Local-only like `qwenPreserveThinking`: first-party Qwen APIs
|
||||
// (Dashscope, Qwen Portal) drive effort through their own OpenAI-style
|
||||
// dialect, and local Ollama keeps its native effort vocabulary.
|
||||
qwenTemplateReasoningEffort:
|
||||
(thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") &&
|
||||
isLocalOpenAICompatBackend &&
|
||||
provider !== "ollama" &&
|
||||
isQwen38PlusTemplateEffortModelId(spec.id),
|
||||
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
|
||||
cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined,
|
||||
supportsPromptCacheBreakpoints,
|
||||
@@ -761,6 +774,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
// Responses-only; the Qwen `preserve_thinking` template knob lives on
|
||||
// the chat-completions wire shape, never on Responses.
|
||||
qwenPreserveThinking: false,
|
||||
qwenTemplateReasoningEffort: false,
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
requiresToolResultName: false,
|
||||
|
||||
@@ -73,6 +73,25 @@ export const isQwenModelId = memo((modelId: string): boolean => {
|
||||
return modelId.toLowerCase().includes("qwen");
|
||||
});
|
||||
|
||||
/**
|
||||
* Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF
|
||||
* names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking
|
||||
* depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`,
|
||||
* template default `xhigh`; thinking itself cannot be disabled). Compared
|
||||
* component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are
|
||||
* excluded — Dashscope drives them through OpenAI-style `reasoning_effort`
|
||||
* with curated compat. The trailing guard rejects parameter-count lookalikes
|
||||
* (`qwen-3.8b`) without breaking `qwen3.8-27b`.
|
||||
*/
|
||||
export const isQwen38PlusTemplateEffortModelId = memo((modelId: string): boolean => {
|
||||
const match = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId);
|
||||
if (!match) return false;
|
||||
const major = Number.parseInt(match[1], 10);
|
||||
const minor = Number.parseInt(match[2], 10);
|
||||
if (major < 3 || (major === 3 && minor < 8)) return false;
|
||||
return !/^-max(?:$|[-.:])/i.test(modelId.slice(match.index + match[0].length));
|
||||
});
|
||||
|
||||
/** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */
|
||||
export const isGemmaModelId = memo((modelId: string): boolean => {
|
||||
return /(^|\/)gemma[-.]?\d/i.test(modelId);
|
||||
|
||||
@@ -71,6 +71,11 @@ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Hi
|
||||
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
|
||||
/** OpenRouter's DeepSeek route accepts only `high`. */
|
||||
const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High];
|
||||
/**
|
||||
* Qwen 3.8+ open-weight chat template: prompt-steered `reasoning_effort`
|
||||
* kwarg with exactly three wire tiers (template default is `xhigh`).
|
||||
*/
|
||||
const QWEN38_TEMPLATE_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.XHigh];
|
||||
/**
|
||||
* Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models
|
||||
* with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the
|
||||
@@ -179,7 +184,9 @@ function fillThinkingWireDefaults<TApi extends Api>(
|
||||
thinking.supportsDisplay === undefined &&
|
||||
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
||||
supportsAdaptiveThinkingDisplay(spec.id);
|
||||
const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
|
||||
const needsRequiresEffort =
|
||||
thinking.requiresEffort === undefined &&
|
||||
(impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat));
|
||||
const needsDefaultLevel =
|
||||
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
|
||||
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
|
||||
@@ -232,7 +239,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
|
||||
) {
|
||||
config.supportsDisplay = true;
|
||||
}
|
||||
if (impliesMandatoryReasoning(parsed, spec.id)) {
|
||||
if (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)) {
|
||||
config.requiresEffort = true;
|
||||
}
|
||||
return config;
|
||||
@@ -376,6 +383,13 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
if (spec.provider === "ollama") {
|
||||
return OLLAMA_REASONING_EFFORTS;
|
||||
}
|
||||
// Qwen 3.8+ served through a local llama.cpp-style backend: the chat
|
||||
// template's prompt-steered `reasoning_effort` kwarg accepts exactly
|
||||
// low/medium/xhigh (and thinking cannot be turned off — the official 3.8
|
||||
// template raises on `enable_thinking: false`, hence requiresEffort).
|
||||
if (isOpenAICompatReasoningApi(spec.api) && isQwenTemplateReasoningEffortCompat(compat)) {
|
||||
return QWEN38_TEMPLATE_REASONING_EFFORTS;
|
||||
}
|
||||
if (
|
||||
(isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
|
||||
isDeepseekReasoningModel(spec)
|
||||
@@ -483,6 +497,12 @@ function isOpenRouterThinkingFormat(compat: CompatOf<Api>): boolean {
|
||||
function isZaiThinkingFormat(compat: CompatOf<Api>): boolean {
|
||||
return compat !== undefined && "thinkingFormat" in compat && compat.thinkingFormat === "zai";
|
||||
}
|
||||
/** Resolved-compat gate for the Qwen 3.8+ local template `reasoning_effort` dialect. */
|
||||
function isQwenTemplateReasoningEffortCompat(compat: CompatOf<Api>): boolean {
|
||||
return (
|
||||
compat !== undefined && "qwenTemplateReasoningEffort" in compat && compat.qwenTemplateReasoningEffort === true
|
||||
);
|
||||
}
|
||||
|
||||
function inferDetectedEffortMap<TApi extends Api>(
|
||||
spec: ModelSpec<TApi>,
|
||||
|
||||
@@ -73,8 +73,10 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
||||
case "openrouter":
|
||||
return "openrouter:pseudo-api";
|
||||
case "vllm": {
|
||||
// v2: qwen3.8 rows cached before the reasoning/template-effort upgrade
|
||||
// carry `reasoning: false` and must be refetched.
|
||||
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
|
||||
return `vllm:${Bun.hash(baseUrl).toString(36)}`;
|
||||
return `vllm:models-v2:${Bun.hash(baseUrl).toString(36)}`;
|
||||
}
|
||||
default:
|
||||
return providerId;
|
||||
|
||||
@@ -16,6 +16,7 @@ import {
|
||||
isGrokReasoningEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isKimiModelId,
|
||||
isQwen38PlusTemplateEffortModelId,
|
||||
isReasoningGlmModelId,
|
||||
} from "../identity/family";
|
||||
import { resolveModelReference } from "../identity/reference";
|
||||
@@ -4994,6 +4995,11 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
|
||||
return {
|
||||
...model,
|
||||
contextWindow: toPositiveNumber(entry.max_model_len, model.contextWindow),
|
||||
// vLLM's /v1/models reports no reasoning capability. Qwen 3.8+
|
||||
// open weights always think (the template cannot disable it), so
|
||||
// light up the effort dial; buildModel derives the template
|
||||
// ladder from the id + local-backend compat.
|
||||
reasoning: model.reasoning || isQwen38PlusTemplateEffortModelId(model.id),
|
||||
};
|
||||
},
|
||||
fetch: config?.fetch,
|
||||
|
||||
@@ -259,6 +259,16 @@ export interface OpenAICompat {
|
||||
* Non-Qwen templates ignore the flag, so the auto-detection is safe.
|
||||
*/
|
||||
qwenPreserveThinking?: boolean;
|
||||
/**
|
||||
* Route the requested thinking effort onto the Qwen 3.8+ chat template's
|
||||
* `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default
|
||||
* `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects
|
||||
* (plus the top-level field on the `qwen` dialect, which newer llama.cpp
|
||||
* builds map natively). Without it the qwen dialects only toggle
|
||||
* `enable_thinking` and the template always thinks at its `xhigh` default.
|
||||
* Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend).
|
||||
*/
|
||||
qwenTemplateReasoningEffort?: boolean;
|
||||
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
|
||||
requiresAssistantContentForToolCalls?: boolean;
|
||||
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
|
||||
@@ -604,6 +614,7 @@ export interface ResolvedOpenAISharedCompat {
|
||||
allowsSyntheticReasoningContentForToolCalls: boolean;
|
||||
replayReasoningContent: boolean;
|
||||
qwenPreserveThinking: boolean;
|
||||
qwenTemplateReasoningEffort: boolean;
|
||||
requiresThinkingAsText: boolean;
|
||||
requiresMistralToolIds: boolean;
|
||||
requiresToolResultName: boolean;
|
||||
@@ -669,6 +680,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
||||
| "allowsSyntheticReasoningContentForToolCalls"
|
||||
| "replayReasoningContent"
|
||||
| "qwenPreserveThinking"
|
||||
| "qwenTemplateReasoningEffort"
|
||||
| "requiresThinkingAsText"
|
||||
| "requiresMistralToolIds"
|
||||
| "requiresToolResultName"
|
||||
|
||||
@@ -14,6 +14,7 @@ import {
|
||||
isMinimaxM3FamilyModelId,
|
||||
isOpenAIGptOssModelId,
|
||||
isOpenAIModelId,
|
||||
isQwen38PlusTemplateEffortModelId,
|
||||
isReasoningGlmModelId,
|
||||
modelFamilyToken,
|
||||
parseAnthropicModel,
|
||||
@@ -30,6 +31,27 @@ describe("isKimiModelId", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("isQwen38PlusTemplateEffortModelId", () => {
|
||||
test("matches Qwen 3.8+ open-weight ids across id shapes and versions", () => {
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-2.4t-a95b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen/qwen3.8-27b")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("Qwen3.8-27B-UD-Q6_K_XL")).toBe(true);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b:thinking")).toBe(true);
|
||||
// Component-wise version compare: 3.10 sorts after 3.8.
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.10-27b")).toBe(true);
|
||||
});
|
||||
test("rejects pre-3.8 versions, parameter-count lookalikes, and API-only Max SKUs", () => {
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3-8b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen-3.6-27b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.7-plus")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen2.5-coder-7b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen-3.8b")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max")).toBe(false);
|
||||
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max-preview")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isKimiK26ModelId", () => {
|
||||
test("matches Kimi K2.6 without accepting adjacent versions", () => {
|
||||
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
|
||||
|
||||
@@ -984,3 +984,87 @@ describe("model thinking runtime helpers", () => {
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("Qwen 3.8 local template effort ladder", () => {
|
||||
it("derives the low/medium/xhigh ladder with mandatory effort on local llama.cpp-style backends", () => {
|
||||
const llamaCpp = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "llama.cpp",
|
||||
baseUrl: "http://127.0.0.1:8080/v1",
|
||||
});
|
||||
// Official 3.8 template: reasoning_effort accepts exactly low/medium/xhigh
|
||||
// and raises on `enable_thinking: false` — off must clamp, never disable.
|
||||
expect(llamaCpp.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
});
|
||||
expect(llamaCpp.compat.qwenTemplateReasoningEffort).toBe(true);
|
||||
// Unsupported tiers clamp onto real wire tiers: high floors to medium
|
||||
// (xhigh is a deliberate opt-in), minimal floors to low.
|
||||
expect(clampThinkingLevelForModel(llamaCpp, Effort.High)).toBe(Effort.Medium);
|
||||
expect(clampThinkingLevelForModel(llamaCpp, Effort.Minimal)).toBe(Effort.Low);
|
||||
expect(minimumSupportedEffort(llamaCpp)).toBe(Effort.Low);
|
||||
});
|
||||
|
||||
it("normalizes a stale cached generic ladder to the template ladder", () => {
|
||||
const cached = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
baseUrl: "http://127.0.0.1:8000/v1",
|
||||
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
});
|
||||
expect(cached.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("routes vLLM Qwen through the chat_template_kwargs dialect", () => {
|
||||
// vLLM ignores top-level `enable_thinking`; only chat_template_kwargs
|
||||
// reach the template renderer.
|
||||
const vllm = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
baseUrl: "http://127.0.0.1:8000/v1",
|
||||
});
|
||||
expect(vllm.compat.thinkingFormat).toBe("qwen-chat-template");
|
||||
expect(vllm.compat.reasoningDisableMode).toBe("qwen-template-false");
|
||||
expect(vllm.compat.qwenTemplateReasoningEffort).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps hosted, pre-3.8, and local-Ollama Qwen off the template ladder", () => {
|
||||
const hosted = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "nanogpt",
|
||||
baseUrl: "https://nano-gpt.com/api/v1",
|
||||
});
|
||||
expect(hosted.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(hosted.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
|
||||
|
||||
const qwen36 = createModel({
|
||||
id: "qwen-3.6-27b",
|
||||
api: "openai-completions",
|
||||
provider: "llama.cpp",
|
||||
baseUrl: "http://localhost:8080/v1",
|
||||
});
|
||||
expect(qwen36.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(qwen36.thinking?.requiresEffort).toBeUndefined();
|
||||
|
||||
// Local Ollama renders its own (Go) templates and keeps the native
|
||||
// low/medium/high/max effort vocabulary.
|
||||
const ollama = createModel({
|
||||
id: "qwen3.8-27b",
|
||||
api: "openai-completions",
|
||||
provider: "ollama",
|
||||
baseUrl: "http://127.0.0.1:11434/v1",
|
||||
});
|
||||
expect(ollama.compat.qwenTemplateReasoningEffort).toBe(false);
|
||||
expect(ollama.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { vllmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
describe("vLLM provider discovery", () => {
|
||||
test("lights up the reasoning dial for Qwen 3.8+ despite silent /v1/models metadata", async () => {
|
||||
// vLLM's /v1/models never advertises reasoning; without the id-based
|
||||
// upgrade a served Qwen3.8 loses its effort dial entirely and always
|
||||
// thinks at the template's xhigh default.
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{ id: "qwen3.8-27b", object: "model", max_model_len: 262144 },
|
||||
{ id: "qwen2.5-coder-7b", object: "model", max_model_len: 131072 },
|
||||
],
|
||||
}),
|
||||
{ status: 200, headers: { "content-type": "application/json" } },
|
||||
);
|
||||
|
||||
const options = vllmModelManagerOptions({ fetch: fetchMock });
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(models?.find(model => model.id === "qwen3.8-27b")).toMatchObject({
|
||||
provider: "vllm",
|
||||
api: "openai-completions",
|
||||
reasoning: true,
|
||||
contextWindow: 262144,
|
||||
});
|
||||
// Non-thinking Qwen generations keep the wire-reported default.
|
||||
expect(models?.find(model => model.id === "qwen2.5-coder-7b")?.reasoning).toBe(false);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user