From bf490ae024db3d447ba99eedc46e18fc381bf10c Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 19 Aug 2026 00:47:11 +0200 Subject: [PATCH] fix: added reasoning effort support for qwen templates - Added `reasoning_effort` kwarg and top-level support for Qwen 3.8+ templates. - Introduced `qwenTemplateReasoningEffort` compatibility option and identity helpers. - Enabled default reasoning enforcement and updated cache provider invalidation. - Added comprehensive unit and compatibility test suites for Qwen reasoning dials. --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/openai-shared.ts | 26 +++++- .../ai/test/issue-967-vision-guard.test.ts | 1 + packages/ai/test/openai-compat-policy.test.ts | 66 +++++++++++++++ .../ai/test/openai-completions-compat.test.ts | 1 + ...nai-completions-tool-result-images.test.ts | 1 + packages/catalog/CHANGELOG.md | 4 + packages/catalog/src/compat/openai.ts | 16 +++- packages/catalog/src/identity/family.ts | 19 +++++ packages/catalog/src/model-thinking.ts | 24 +++++- .../src/provider-models/cache-provider-id.ts | 4 +- .../src/provider-models/openai-compat.ts | 6 ++ packages/catalog/src/types.ts | 12 +++ packages/catalog/test/identity-family.test.ts | 22 +++++ packages/catalog/test/model-thinking.test.ts | 84 +++++++++++++++++++ packages/catalog/test/vllm-provider.test.ts | 33 ++++++++ 16 files changed, 317 insertions(+), 6 deletions(-) create mode 100644 packages/catalog/test/vllm-provider.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e469ca4f6..544910b87 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed thinking effort selections being ignored for local Qwen 3.8+ models on llama.cpp and vLLM: the Qwen chat-completions dialects only toggled `enable_thinking`, so the chat template always reasoned at its `xhigh` default no matter which level was selected. The encoder now routes the requested effort onto the template's `reasoning_effort` kwarg (`chat_template_kwargs` for both Qwen dialects, plus the top-level field newer llama.cpp builds map natively). + ## [17.3.7] - 2026-08-17 ### Changed diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 34a4bbae0..5e7e96321 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -747,7 +747,7 @@ export type OpenAICompletionsParams = Omit { expect(params.enable_thinking).toBe(true); expect(params.reasoning_effort).toBeUndefined(); }); + + function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> { + return buildModel({ + id, + name: id, + api: "openai-completions", + provider, + baseUrl, + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262_144, + maxTokens: 32_768, + } satisfies ModelSpec<"openai-completions">); + } + + it("routes local Qwen3.8 effort selections onto the chat template (llama.cpp qwen dialect)", () => { + // Regression: the qwen dialects used to emit only `enable_thinking: true`, + // so every effort selection ran at the template's xhigh default. + const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1"); + for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) { + const params = chatParams(); + applyChatCompletionsCompatPolicy( + params, + resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }), + ); + // Twin emission: top-level for newer llama.cpp builds, kwargs for + // older builds — and the preserve_thinking kwarg must survive. + expect(params.enable_thinking).toBe(true); + expect(params.reasoning_effort).toBe(effort); + expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: effort }); + expect(params.preserve_thinking).toBe(true); + } + }); + + it("routes local Qwen3.8 effort selections via chat_template_kwargs only on vLLM", () => { + // vLLM's renderer reads chat_template_kwargs; NIM-style schemas reject + // unknown top-level fields, so nothing may ride top-level here. + const model = localQwenModel("qwen3.8-27b", "vllm", "http://127.0.0.1:8000/v1"); + const params = chatParams(); + applyChatCompletionsCompatPolicy( + params, + resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }), + ); + expect(params.enable_thinking).toBeUndefined(); + expect(params.reasoning_effort).toBeUndefined(); + expect(params.chat_template_kwargs).toEqual({ + preserve_thinking: true, + enable_thinking: true, + reasoning_effort: Effort.Medium, + }); + }); + + it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => { + // Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would + // inject an undefined template variable for zero benefit. + const model = localQwenModel("qwen-3.6-27b", "llama.cpp", "http://127.0.0.1:8080/v1"); + const params = chatParams(); + applyChatCompletionsCompatPolicy( + params, + resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.High }), + ); + expect(params.enable_thinking).toBe(true); + expect(params.reasoning_effort).toBeUndefined(); + expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true }); + }); }); diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 12b679793..defa14816 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -202,6 +202,7 @@ describe("openai-completions compatibility", () => { allowsSyntheticReasoningContentForToolCalls: true, replayReasoningContent: false, qwenPreserveThinking: false, + qwenTemplateReasoningEffort: false, requiresAssistantContentForToolCalls: false, openRouterRouting: {}, vercelGatewayRouting: {}, diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index 18ef096d8..083022e83 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -43,6 +43,7 @@ const compat: ResolvedOpenAICompat = { allowsSyntheticReasoningContentForToolCalls: true, replayReasoningContent: false, qwenPreserveThinking: false, + qwenTemplateReasoningEffort: false, requiresAssistantContentForToolCalls: false, openRouterRouting: {}, vercelGatewayRouting: {}, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 288b1940c..21ffab449 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning. + ## [17.3.6] - 2026-08-17 ### Changed diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 2eeec29f1..15f45f9eb 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -22,6 +22,7 @@ import { isKimiModelId, isMimoModelIdOrName, isOpenAISamplingRestrictedModelId, + isQwen38PlusTemplateEffortModelId, isQwenModelId, } from "../identity/family"; import type { @@ -464,7 +465,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv ? "zai" : isOpenRouter ? "openrouter" - : isQwen && isNvidiaNim + : isQwen && (isNvidiaNim || provider === "vllm") ? "qwen-chat-template" : isQwen && isFireworks ? "openai" @@ -587,6 +588,18 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // parameter, so the flag stays a no-op outside the Qwen path. qwenPreserveThinking: (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && isLocalOpenAICompatBackend, + // Qwen 3.8+ templates steer thinking depth via the `reasoning_effort` + // template kwarg (low/medium/xhigh, default xhigh); without routing the + // requested effort there, the enable_thinking toggle alone leaves the + // model at xhigh no matter what the user selects. + // Local-only like `qwenPreserveThinking`: first-party Qwen APIs + // (Dashscope, Qwen Portal) drive effort through their own OpenAI-style + // dialect, and local Ollama keeps its native effort vocabulary. + qwenTemplateReasoningEffort: + (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && + isLocalOpenAICompatBackend && + provider !== "ollama" && + isQwen38PlusTemplateEffortModelId(spec.id), requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning, cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined, supportsPromptCacheBreakpoints, @@ -761,6 +774,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol // Responses-only; the Qwen `preserve_thinking` template knob lives on // the chat-completions wire shape, never on Responses. qwenPreserveThinking: false, + qwenTemplateReasoningEffort: false, requiresThinkingAsText: false, requiresMistralToolIds: false, requiresToolResultName: false, diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index 9395c3e13..6bb563ca5 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -73,6 +73,25 @@ export const isQwenModelId = memo((modelId: string): boolean => { return modelId.toLowerCase().includes("qwen"); }); +/** + * Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF + * names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking + * depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`, + * template default `xhigh`; thinking itself cannot be disabled). Compared + * component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are + * excluded — Dashscope drives them through OpenAI-style `reasoning_effort` + * with curated compat. The trailing guard rejects parameter-count lookalikes + * (`qwen-3.8b`) without breaking `qwen3.8-27b`. + */ +export const isQwen38PlusTemplateEffortModelId = memo((modelId: string): boolean => { + const match = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId); + if (!match) return false; + const major = Number.parseInt(match[1], 10); + const minor = Number.parseInt(match[2], 10); + if (major < 3 || (major === 3 && minor < 8)) return false; + return !/^-max(?:$|[-.:])/i.test(modelId.slice(match.index + match[0].length)); +}); + /** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */ export const isGemmaModelId = memo((modelId: string): boolean => { return /(^|\/)gemma[-.]?\d/i.test(modelId); diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 6d0c40428..3f89ae195 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -71,6 +71,11 @@ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Hi const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max]; /** OpenRouter's DeepSeek route accepts only `high`. */ const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High]; +/** + * Qwen 3.8+ open-weight chat template: prompt-steered `reasoning_effort` + * kwarg with exactly three wire tiers (template default is `xhigh`). + */ +const QWEN38_TEMPLATE_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.XHigh]; /** * Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models * with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the @@ -179,7 +184,9 @@ function fillThinkingWireDefaults( thinking.supportsDisplay === undefined && (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && supportsAdaptiveThinkingDisplay(spec.id); - const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id); + const needsRequiresEffort = + thinking.requiresEffort === undefined && + (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)); const needsDefaultLevel = thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)); if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) { @@ -232,7 +239,7 @@ export function deriveThinking(spec: ModelSpec, compat: ) { config.supportsDisplay = true; } - if (impliesMandatoryReasoning(parsed, spec.id)) { + if (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)) { config.requiresEffort = true; } return config; @@ -376,6 +383,13 @@ function getModelDefinedEfforts( if (spec.provider === "ollama") { return OLLAMA_REASONING_EFFORTS; } + // Qwen 3.8+ served through a local llama.cpp-style backend: the chat + // template's prompt-steered `reasoning_effort` kwarg accepts exactly + // low/medium/xhigh (and thinking cannot be turned off — the official 3.8 + // template raises on `enable_thinking: false`, hence requiresEffort). + if (isOpenAICompatReasoningApi(spec.api) && isQwenTemplateReasoningEffortCompat(compat)) { + return QWEN38_TEMPLATE_REASONING_EFFORTS; + } if ( (isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) && isDeepseekReasoningModel(spec) @@ -483,6 +497,12 @@ function isOpenRouterThinkingFormat(compat: CompatOf): boolean { function isZaiThinkingFormat(compat: CompatOf): boolean { return compat !== undefined && "thinkingFormat" in compat && compat.thinkingFormat === "zai"; } +/** Resolved-compat gate for the Qwen 3.8+ local template `reasoning_effort` dialect. */ +function isQwenTemplateReasoningEffortCompat(compat: CompatOf): boolean { + return ( + compat !== undefined && "qwenTemplateReasoningEffort" in compat && compat.qwenTemplateReasoningEffort === true + ); +} function inferDetectedEffortMap( spec: ModelSpec, diff --git a/packages/catalog/src/provider-models/cache-provider-id.ts b/packages/catalog/src/provider-models/cache-provider-id.ts index 3c5032e9c..7e95fa3fb 100644 --- a/packages/catalog/src/provider-models/cache-provider-id.ts +++ b/packages/catalog/src/provider-models/cache-provider-id.ts @@ -73,8 +73,10 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa case "openrouter": return "openrouter:pseudo-api"; case "vllm": { + // v2: qwen3.8 rows cached before the reasoning/template-effort upgrade + // carry `reasoning: false` and must be refetched. const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!; - return `vllm:${Bun.hash(baseUrl).toString(36)}`; + return `vllm:models-v2:${Bun.hash(baseUrl).toString(36)}`; } default: return providerId; diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 8ade9a049..503fd7c8b 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -16,6 +16,7 @@ import { isGrokReasoningEffortCapable, isKimiK3ModelId, isKimiModelId, + isQwen38PlusTemplateEffortModelId, isReasoningGlmModelId, } from "../identity/family"; import { resolveModelReference } from "../identity/reference"; @@ -4994,6 +4995,11 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM return { ...model, contextWindow: toPositiveNumber(entry.max_model_len, model.contextWindow), + // vLLM's /v1/models reports no reasoning capability. Qwen 3.8+ + // open weights always think (the template cannot disable it), so + // light up the effort dial; buildModel derives the template + // ladder from the id + local-backend compat. + reasoning: model.reasoning || isQwen38PlusTemplateEffortModelId(model.id), }; }, fetch: config?.fetch, diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 5760208f9..3ef05e849 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -259,6 +259,16 @@ export interface OpenAICompat { * Non-Qwen templates ignore the flag, so the auto-detection is safe. */ qwenPreserveThinking?: boolean; + /** + * Route the requested thinking effort onto the Qwen 3.8+ chat template's + * `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default + * `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects + * (plus the top-level field on the `qwen` dialect, which newer llama.cpp + * builds map natively). Without it the qwen dialects only toggle + * `enable_thinking` and the template always thinks at its `xhigh` default. + * Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend). + */ + qwenTemplateReasoningEffort?: boolean; /** Whether assistant tool-call messages must include non-empty content. Default: false. */ requiresAssistantContentForToolCalls?: boolean; /** Whether the provider supports the `tool_choice` parameter. Default: true. */ @@ -604,6 +614,7 @@ export interface ResolvedOpenAISharedCompat { allowsSyntheticReasoningContentForToolCalls: boolean; replayReasoningContent: boolean; qwenPreserveThinking: boolean; + qwenTemplateReasoningEffort: boolean; requiresThinkingAsText: boolean; requiresMistralToolIds: boolean; requiresToolResultName: boolean; @@ -669,6 +680,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & | "allowsSyntheticReasoningContentForToolCalls" | "replayReasoningContent" | "qwenPreserveThinking" + | "qwenTemplateReasoningEffort" | "requiresThinkingAsText" | "requiresMistralToolIds" | "requiresToolResultName" diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index 52b57dedd..101ffc77f 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -14,6 +14,7 @@ import { isMinimaxM3FamilyModelId, isOpenAIGptOssModelId, isOpenAIModelId, + isQwen38PlusTemplateEffortModelId, isReasoningGlmModelId, modelFamilyToken, parseAnthropicModel, @@ -30,6 +31,27 @@ describe("isKimiModelId", () => { }); }); +describe("isQwen38PlusTemplateEffortModelId", () => { + test("matches Qwen 3.8+ open-weight ids across id shapes and versions", () => { + expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b")).toBe(true); + expect(isQwen38PlusTemplateEffortModelId("qwen3.8-2.4t-a95b")).toBe(true); + expect(isQwen38PlusTemplateEffortModelId("qwen/qwen3.8-27b")).toBe(true); + expect(isQwen38PlusTemplateEffortModelId("Qwen3.8-27B-UD-Q6_K_XL")).toBe(true); + expect(isQwen38PlusTemplateEffortModelId("qwen3.8-27b:thinking")).toBe(true); + // Component-wise version compare: 3.10 sorts after 3.8. + expect(isQwen38PlusTemplateEffortModelId("qwen3.10-27b")).toBe(true); + }); + test("rejects pre-3.8 versions, parameter-count lookalikes, and API-only Max SKUs", () => { + expect(isQwen38PlusTemplateEffortModelId("qwen3-8b")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen-3.6-27b")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen3.7-plus")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen2.5-coder-7b")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen-3.8b")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max")).toBe(false); + expect(isQwen38PlusTemplateEffortModelId("qwen3.8-max-preview")).toBe(false); + }); +}); + describe("isKimiK26ModelId", () => { test("matches Kimi K2.6 without accepting adjacent versions", () => { expect(isKimiK26ModelId("kimi-k2.6")).toBe(true); diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index e55e767c0..41b15e7b2 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -984,3 +984,87 @@ describe("model thinking runtime helpers", () => { }); }); }); + +describe("Qwen 3.8 local template effort ladder", () => { + it("derives the low/medium/xhigh ladder with mandatory effort on local llama.cpp-style backends", () => { + const llamaCpp = createModel({ + id: "qwen3.8-27b", + api: "openai-completions", + provider: "llama.cpp", + baseUrl: "http://127.0.0.1:8080/v1", + }); + // Official 3.8 template: reasoning_effort accepts exactly low/medium/xhigh + // and raises on `enable_thinking: false` — off must clamp, never disable. + expect(llamaCpp.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.XHigh], + requiresEffort: true, + }); + expect(llamaCpp.compat.qwenTemplateReasoningEffort).toBe(true); + // Unsupported tiers clamp onto real wire tiers: high floors to medium + // (xhigh is a deliberate opt-in), minimal floors to low. + expect(clampThinkingLevelForModel(llamaCpp, Effort.High)).toBe(Effort.Medium); + expect(clampThinkingLevelForModel(llamaCpp, Effort.Minimal)).toBe(Effort.Low); + expect(minimumSupportedEffort(llamaCpp)).toBe(Effort.Low); + }); + + it("normalizes a stale cached generic ladder to the template ladder", () => { + const cached = createModel({ + id: "qwen3.8-27b", + api: "openai-completions", + provider: "vllm", + baseUrl: "http://127.0.0.1:8000/v1", + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + }); + expect(cached.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.XHigh], + requiresEffort: true, + }); + }); + + it("routes vLLM Qwen through the chat_template_kwargs dialect", () => { + // vLLM ignores top-level `enable_thinking`; only chat_template_kwargs + // reach the template renderer. + const vllm = createModel({ + id: "qwen3.8-27b", + api: "openai-completions", + provider: "vllm", + baseUrl: "http://127.0.0.1:8000/v1", + }); + expect(vllm.compat.thinkingFormat).toBe("qwen-chat-template"); + expect(vllm.compat.reasoningDisableMode).toBe("qwen-template-false"); + expect(vllm.compat.qwenTemplateReasoningEffort).toBe(true); + }); + + it("keeps hosted, pre-3.8, and local-Ollama Qwen off the template ladder", () => { + const hosted = createModel({ + id: "qwen3.8-27b", + api: "openai-completions", + provider: "nanogpt", + baseUrl: "https://nano-gpt.com/api/v1", + }); + expect(hosted.compat.qwenTemplateReasoningEffort).toBe(false); + expect(hosted.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); + + const qwen36 = createModel({ + id: "qwen-3.6-27b", + api: "openai-completions", + provider: "llama.cpp", + baseUrl: "http://localhost:8080/v1", + }); + expect(qwen36.compat.qwenTemplateReasoningEffort).toBe(false); + expect(qwen36.thinking?.requiresEffort).toBeUndefined(); + + // Local Ollama renders its own (Go) templates and keeps the native + // low/medium/high/max effort vocabulary. + const ollama = createModel({ + id: "qwen3.8-27b", + api: "openai-completions", + provider: "ollama", + baseUrl: "http://127.0.0.1:11434/v1", + }); + expect(ollama.compat.qwenTemplateReasoningEffort).toBe(false); + expect(ollama.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + }); +}); diff --git a/packages/catalog/test/vllm-provider.test.ts b/packages/catalog/test/vllm-provider.test.ts new file mode 100644 index 000000000..ad22feaa6 --- /dev/null +++ b/packages/catalog/test/vllm-provider.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, test } from "bun:test"; +import { vllmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +describe("vLLM provider discovery", () => { + test("lights up the reasoning dial for Qwen 3.8+ despite silent /v1/models metadata", async () => { + // vLLM's /v1/models never advertises reasoning; without the id-based + // upgrade a served Qwen3.8 loses its effort dial entirely and always + // thinks at the template's xhigh default. + const fetchMock: FetchImpl = async () => + new Response( + JSON.stringify({ + data: [ + { id: "qwen3.8-27b", object: "model", max_model_len: 262144 }, + { id: "qwen2.5-coder-7b", object: "model", max_model_len: 131072 }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + + const options = vllmModelManagerOptions({ fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + + expect(models?.find(model => model.id === "qwen3.8-27b")).toMatchObject({ + provider: "vllm", + api: "openai-completions", + reasoning: true, + contextWindow: 262144, + }); + // Non-thinking Qwen generations keep the wire-reported default. + expect(models?.find(model => model.id === "qwen2.5-coder-7b")?.reasoning).toBe(false); + }); +});