fix: added reasoning effort support for qwen templates

- Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates.
- Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers.
- Enabled default reasoning enforcement and updated cache provider invalidation.
- Added comprehensive unit and compatibility test suites for Qwen reasoning dials.
This commit is contained in:
can1357
2026-08-19 00:47:11 +02:00
parent 8500092296
commit bf490ae024
16 changed files with 317 additions and 6 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed thinking effort selections being ignored for local Qwen 3.8+ models on llama.cpp and vLLM: the Qwen chat-completions dialects only toggled `enable_thinking`, so the chat template always reasoned at its `xhigh` default no matter which level was selected. The encoder now routes the requested effort onto the template's `reasoning_effort` kwarg (`chat_template_kwargs` for both Qwen dialects, plus the top-level field newer llama.cpp builds map natively).
## [17.3.7] - 2026-08-17
### Changed
+24 -2
View File
@@ -747,7 +747,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
thinking?: { type: "enabled" | "disabled"; effort?: string; keep?: "all" };
enable_thinking?: boolean;
preserve_thinking?: boolean;
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean; reasoning_effort?: string };
reasoning?: { effort?: string } | { enabled: false };
reasoning_effort?: string | null;
service_tier?: ServiceTier;
@@ -1049,12 +1049,34 @@ export function applyChatCompletionsCompatPolicy(params: OpenAICompletionsParams
break;
case "qwen-enable-thinking-false":
params.enable_thinking = true;
// Qwen 3.8+ templates steer thinking depth via the
// `reasoning_effort` kwarg (low/medium/xhigh, template default
// xhigh) — without it every effort selection lands on xhigh.
// Twin emission mirrors `preserve_thinking` above: newer
// llama.cpp builds map the top-level OpenAI field into the
// template, older builds and Alibaba-style local servers read
// only the kwargs copy. The `qwen-chat-template` dialect (NIM,
// vLLM/SGLang) rides kwargs alone — NIM's request schema
// rejects unknown top-level fields (#2299).
if (policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined) {
params.reasoning_effort = reasoning.wireEffort;
params.chat_template_kwargs = {
...params.chat_template_kwargs,
reasoning_effort: reasoning.wireEffort,
};
}
break;
case "qwen-template-false":
// Spread so the `preserve_thinking` kwarg hoisted above
// survives the merge — a bare `{ enable_thinking: true }`
// would clobber it.
params.chat_template_kwargs = { ...params.chat_template_kwargs, enable_thinking: true };
params.chat_template_kwargs = {
...params.chat_template_kwargs,
enable_thinking: true,
...(policy.compat.qwenTemplateReasoningEffort && reasoning.wireEffort !== undefined
? { reasoning_effort: reasoning.wireEffort }
: {}),
};
break;
case "openrouter-enabled-false":
if (reasoning.wireEffort !== undefined) {
@@ -55,6 +55,7 @@ const compat: ResolvedOpenAICompat = {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
@@ -199,4 +199,70 @@ describe("OpenAI compat policy", () => {
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBeUndefined();
});
function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> {
return buildModel({
id,
name: id,
api: "openai-completions",
provider,
baseUrl,
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 262_144,
maxTokens: 32_768,
} satisfies ModelSpec<"openai-completions">);
}
it("routes local Qwen3.8 effort selections onto the chat template (llama.cpp qwen dialect)", () => {
// Regression: the qwen dialects used to emit only `enable_thinking: true`,
// so every effort selection ran at the template's xhigh default.
const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
);
// Twin emission: top-level for newer llama.cpp builds, kwargs for
// older builds — and the preserve_thinking kwarg must survive.
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBe(effort);
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: effort });
expect(params.preserve_thinking).toBe(true);
}
});
it("routes local Qwen3.8 effort selections via chat_template_kwargs only on vLLM", () => {
// vLLM's renderer reads chat_template_kwargs; NIM-style schemas reject
// unknown top-level fields, so nothing may ride top-level here.
const model = localQwenModel("qwen3.8-27b", "vllm", "http://127.0.0.1:8000/v1");
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }),
);
expect(params.enable_thinking).toBeUndefined();
expect(params.reasoning_effort).toBeUndefined();
expect(params.chat_template_kwargs).toEqual({
preserve_thinking: true,
enable_thinking: true,
reasoning_effort: Effort.Medium,
});
});
it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => {
// Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would
// inject an undefined template variable for zero benefit.
const model = localQwenModel("qwen-3.6-27b", "llama.cpp", "http://127.0.0.1:8080/v1");
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.High }),
);
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBeUndefined();
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
});
});
@@ -202,6 +202,7 @@ describe("openai-completions compatibility", () => {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
@@ -43,6 +43,7 @@ const compat: ResolvedOpenAICompat = {
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning.
## [17.3.6] - 2026-08-17
### Changed
+15 -1
View File
@@ -22,6 +22,7 @@ import {
isKimiModelId,
isMimoModelIdOrName,
isOpenAISamplingRestrictedModelId,
isQwen38PlusTemplateEffortModelId,
isQwenModelId,
} from "../identity/family";
import type {
@@ -464,7 +465,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
? "zai"
: isOpenRouter
? "openrouter"
: isQwen && isNvidiaNim
: isQwen && (isNvidiaNim || provider === "vllm")
? "qwen-chat-template"
: isQwen && isFireworks
? "openai"
@@ -587,6 +588,18 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// parameter, so the flag stays a no-op outside the Qwen path.
qwenPreserveThinking:
(thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && isLocalOpenAICompatBackend,
// Qwen 3.8+ templates steer thinking depth via the `reasoning_effort`
// template kwarg (low/medium/xhigh, default xhigh); without routing the
// requested effort there, the enable_thinking toggle alone leaves the
// model at xhigh no matter what the user selects.
// Local-only like `qwenPreserveThinking`: first-party Qwen APIs
// (Dashscope, Qwen Portal) drive effort through their own OpenAI-style
// dialect, and local Ollama keeps its native effort vocabulary.
qwenTemplateReasoningEffort:
(thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") &&
isLocalOpenAICompatBackend &&
provider !== "ollama" &&
isQwen38PlusTemplateEffortModelId(spec.id),
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined,
supportsPromptCacheBreakpoints,
@@ -761,6 +774,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
// Responses-only; the Qwen `preserve_thinking` template knob lives on
// the chat-completions wire shape, never on Responses.
qwenPreserveThinking: false,
qwenTemplateReasoningEffort: false,
requiresThinkingAsText: false,
requiresMistralToolIds: false,
requiresToolResultName: false,
+19
View File
@@ -73,6 +73,25 @@ export const isQwenModelId = memo((modelId: string): boolean => {
return modelId.toLowerCase().includes("qwen");
});
/**
* Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF
* names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking
* depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`,
* template default `xhigh`; thinking itself cannot be disabled). Compared
* component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are
* excluded — Dashscope drives them through OpenAI-style `reasoning_effort`
* with curated compat. The trailing guard rejects parameter-count lookalikes
* (`qwen-3.8b`) without breaking `qwen3.8-27b`.
*/
export const isQwen38PlusTemplateEffortModelId = memo((modelId: string): boolean => {
const match = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId);
if (!match) return false;
const major = Number.parseInt(match[1], 10);
const minor = Number.parseInt(match[2], 10);
if (major < 3 || (major === 3 && minor < 8)) return false;
return !/^-max(?:$|[-.:])/i.test(modelId.slice(match.index + match[0].length));
});
/** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */
export const isGemmaModelId = memo((modelId: string): boolean => {
return /(^|\/)gemma[-.]?\d/i.test(modelId);
+22 -2
View File
@@ -71,6 +71,11 @@ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Hi
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
/** OpenRouter's DeepSeek route accepts only `high`. */
const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High];
/**
* Qwen 3.8+ open-weight chat template: prompt-steered `reasoning_effort`
* kwarg with exactly three wire tiers (template default is `xhigh`).
*/
const QWEN38_TEMPLATE_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.XHigh];
/**
* Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models
* with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the
@@ -179,7 +184,9 @@ function fillThinkingWireDefaults<TApi extends Api>(
thinking.supportsDisplay === undefined &&
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
supportsAdaptiveThinkingDisplay(spec.id);
const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
const needsRequiresEffort =
thinking.requiresEffort === undefined &&
(impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat));
const needsDefaultLevel =
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
@@ -232,7 +239,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
) {
config.supportsDisplay = true;
}
if (impliesMandatoryReasoning(parsed, spec.id)) {
if (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)) {
config.requiresEffort = true;
}
return config;
@@ -376,6 +383,13 @@ function getModelDefinedEfforts<TApi extends Api>(
if (spec.provider === "ollama") {
return OLLAMA_REASONING_EFFORTS;
}
// Qwen 3.8+ served through a local llama.cpp-style backend: the chat
// template's prompt-steered `reasoning_effort` kwarg accepts exactly
// low/medium/xhigh (and thinking cannot be turned off — the official 3.8
// template raises on `enable_thinking: false`, hence requiresEffort).
if (isOpenAICompatReasoningApi(spec.api) && isQwenTemplateReasoningEffortCompat(compat)) {
return QWEN38_TEMPLATE_REASONING_EFFORTS;
}
if (
(isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
isDeepseekReasoningModel(spec)
@@ -483,6 +497,12 @@ function isOpenRouterThinkingFormat(compat: CompatOf<Api>): boolean {
function isZaiThinkingFormat(compat: CompatOf<Api>): boolean {
return compat !== undefined && "thinkingFormat" in compat && compat.thinkingFormat === "zai";
}
/** Resolved-compat gate for the Qwen 3.8+ local template `reasoning_effort` dialect. */
function isQwenTemplateReasoningEffortCompat(compat: CompatOf<Api>): boolean {
return (
compat !== undefined && "qwenTemplateReasoningEffort" in compat && compat.qwenTemplateReasoningEffort === true
);
}
function inferDetectedEffortMap<TApi extends Api>(
spec: ModelSpec<TApi>,
@@ -73,8 +73,10 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
case "openrouter":
return "openrouter:pseudo-api";
case "vllm": {
// v2: qwen3.8 rows cached before the reasoning/template-effort upgrade
// carry `reasoning: false` and must be refetched.
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
return `vllm:${Bun.hash(baseUrl).toString(36)}`;
return `vllm:models-v2:${Bun.hash(baseUrl).toString(36)}`;
}
default:
return providerId;
@@ -16,6 +16,7 @@ import {
isGrokReasoningEffortCapable,
isKimiK3ModelId,
isKimiModelId,
isQwen38PlusTemplateEffortModelId,
isReasoningGlmModelId,
} from "../identity/family";
import { resolveModelReference } from "../identity/reference";
@@ -4994,6 +4995,11 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
return {
...model,
contextWindow: toPositiveNumber(entry.max_model_len, model.contextWindow),
// vLLM's /v1/models reports no reasoning capability. Qwen 3.8+
// open weights always think (the template cannot disable it), so
// light up the effort dial; buildModel derives the template
// ladder from the id + local-backend compat.
reasoning: model.reasoning || isQwen38PlusTemplateEffortModelId(model.id),
};
},
fetch: config?.fetch,
+12
View File
@@ -259,6 +259,16 @@ export interface OpenAICompat {
* Non-Qwen templates ignore the flag, so the auto-detection is safe.
*/
qwenPreserveThinking?: boolean;
/**
* Route the requested thinking effort onto the Qwen 3.8+ chat template's
* `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default
* `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects
* (plus the top-level field on the `qwen` dialect, which newer llama.cpp
* builds map natively). Without it the qwen dialects only toggle
* `enable_thinking` and the template always thinks at its `xhigh` default.
* Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend).
*/
qwenTemplateReasoningEffort?: boolean;
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
requiresAssistantContentForToolCalls?: boolean;
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
@@ -604,6 +614,7 @@ export interface ResolvedOpenAISharedCompat {
allowsSyntheticReasoningContentForToolCalls: boolean;
replayReasoningContent: boolean;
qwenPreserveThinking: boolean;
qwenTemplateReasoningEffort: boolean;
requiresThinkingAsText: boolean;
requiresMistralToolIds: boolean;
requiresToolResultName: boolean;
@@ -669,6 +680,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "allowsSyntheticReasoningContentForToolCalls"
| "replayReasoningContent"
| "qwenPreserveThinking"
| "qwenTemplateReasoningEffort"
| "requiresThinkingAsText"
| "requiresMistralToolIds"
| "requiresToolResultName"
@@ -14,6 +14,7 @@ import {
isMinimaxM3FamilyModelId,
isOpenAIGptOssModelId,
isOpenAIModelId,
isQwen38PlusTemplateEffortModelId,
isReasoningGlmModelId,
modelFamilyToken,
parseAnthropicModel,
@@ -30,6 +31,27 @@ describe("isKimiModelId", () => {
});
});
describe("isQwen38PlusTemplateEffortModelId", () => {
test("matches Qwen 3.8+ open-weight ids across id shapes and versions", () => {
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-2.4t-a95b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen/qwen3.8-27b")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("Qwen3.8-27B-UD-Q6_K_XL")).toBe(true);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b:thinking")).toBe(true);
// Component-wise version compare: 3.10 sorts after 3.8.
expect(isQwen38PlusTemplateEffortModelId("qwen3.10-27b")).toBe(true);
});
test("rejects pre-3.8 versions, parameter-count lookalikes, and API-only Max SKUs", () => {
expect(isQwen38PlusTemplateEffortModelId("qwen3-8b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen-3.6-27b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.7-plus")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen2.5-coder-7b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen-3.8b")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max")).toBe(false);
expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max-preview")).toBe(false);
});
});
describe("isKimiK26ModelId", () => {
test("matches Kimi K2.6 without accepting adjacent versions", () => {
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
@@ -984,3 +984,87 @@ describe("model thinking runtime helpers", () => {
});
});
});
describe("Qwen 3.8 local template effort ladder", () => {
it("derives the low/medium/xhigh ladder with mandatory effort on local llama.cpp-style backends", () => {
const llamaCpp = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "llama.cpp",
baseUrl: "http://127.0.0.1:8080/v1",
});
// Official 3.8 template: reasoning_effort accepts exactly low/medium/xhigh
// and raises on `enable_thinking: false` — off must clamp, never disable.
expect(llamaCpp.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
requiresEffort: true,
});
expect(llamaCpp.compat.qwenTemplateReasoningEffort).toBe(true);
// Unsupported tiers clamp onto real wire tiers: high floors to medium
// (xhigh is a deliberate opt-in), minimal floors to low.
expect(clampThinkingLevelForModel(llamaCpp, Effort.High)).toBe(Effort.Medium);
expect(clampThinkingLevelForModel(llamaCpp, Effort.Minimal)).toBe(Effort.Low);
expect(minimumSupportedEffort(llamaCpp)).toBe(Effort.Low);
});
it("normalizes a stale cached generic ladder to the template ladder", () => {
const cached = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "vllm",
baseUrl: "http://127.0.0.1:8000/v1",
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
});
expect(cached.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
requiresEffort: true,
});
});
it("routes vLLM Qwen through the chat_template_kwargs dialect", () => {
// vLLM ignores top-level `enable_thinking`; only chat_template_kwargs
// reach the template renderer.
const vllm = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "vllm",
baseUrl: "http://127.0.0.1:8000/v1",
});
expect(vllm.compat.thinkingFormat).toBe("qwen-chat-template");
expect(vllm.compat.reasoningDisableMode).toBe("qwen-template-false");
expect(vllm.compat.qwenTemplateReasoningEffort).toBe(true);
});
it("keeps hosted, pre-3.8, and local-Ollama Qwen off the template ladder", () => {
const hosted = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "nanogpt",
baseUrl: "https://nano-gpt.com/api/v1",
});
expect(hosted.compat.qwenTemplateReasoningEffort).toBe(false);
expect(hosted.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
const qwen36 = createModel({
id: "qwen-3.6-27b",
api: "openai-completions",
provider: "llama.cpp",
baseUrl: "http://localhost:8080/v1",
});
expect(qwen36.compat.qwenTemplateReasoningEffort).toBe(false);
expect(qwen36.thinking?.requiresEffort).toBeUndefined();
// Local Ollama renders its own (Go) templates and keeps the native
// low/medium/high/max effort vocabulary.
const ollama = createModel({
id: "qwen3.8-27b",
api: "openai-completions",
provider: "ollama",
baseUrl: "http://127.0.0.1:11434/v1",
});
expect(ollama.compat.qwenTemplateReasoningEffort).toBe(false);
expect(ollama.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]);
});
});
@@ -0,0 +1,33 @@
import { describe, expect, test } from "bun:test";
import { vllmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
describe("vLLM provider discovery", () => {
test("lights up the reasoning dial for Qwen 3.8+ despite silent /v1/models metadata", async () => {
// vLLM's /v1/models never advertises reasoning; without the id-based
// upgrade a served Qwen3.8 loses its effort dial entirely and always
// thinks at the template's xhigh default.
const fetchMock: FetchImpl = async () =>
new Response(
JSON.stringify({
data: [
{ id: "qwen3.8-27b", object: "model", max_model_len: 262144 },
{ id: "qwen2.5-coder-7b", object: "model", max_model_len: 131072 },
],
}),
{ status: 200, headers: { "content-type": "application/json" } },
);
const options = vllmModelManagerOptions({ fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
expect(models?.find(model => model.id === "qwen3.8-27b")).toMatchObject({
provider: "vllm",
api: "openai-completions",
reasoning: true,
contextWindow: 262144,
});
// Non-thinking Qwen generations keep the wire-reported default.
expect(models?.find(model => model.id === "qwen2.5-coder-7b")?.reasoning).toBe(false);
});
});