feat(ai): implemented fallback handling and option for qwen reasoning effort

- Added `qwenTemplateReasoningEffort` model compatibility option to disable Qwen chat template kwargs for strict local servers.
- Implemented fallback handling to strip rejected `chat_template_kwargs.reasoning_effort` and hoist values to top-level fields.
- Added comprehensive test coverage for Qwen reasoning effort fallback and keyword rejections.
This commit is contained in:
can1357
2026-08-19 17:40:39 +02:00
parent 7099457f84
commit 74bc1f442e
7 changed files with 267 additions and 5 deletions
+1
View File
@@ -517,6 +517,7 @@ Reasoning / thinking:
- `supportsReasoningParams` — whether request shaping may send reasoning params at all. Default: auto (off for GitHub Copilot chat-completions).
- `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh|max`) to provider-specific strings (e.g. Fireworks GLM maps `minimal -> "none"`).
- `thinkingFormat` — request shape for thinking: `"openai"` (`reasoning_effort`), `"openrouter"` (`reasoning: { effort }`), `"zai"` (`thinking: { type: "enabled" }`), `"qwen"` (top-level `enable_thinking`), or `"qwen-chat-template"` (`chat_template_kwargs.enable_thinking`). Default: `"openai"`.
- `qwenTemplateReasoningEffort` — route the selected effort onto the Qwen 3.8+ chat template's `reasoning_effort` kwarg (`chat_template_kwargs.reasoning_effort`, plus the top-level field on the `qwen` dialect). Default: auto (on for Qwen 3.8+ ids on local non-Ollama backends). Set `false` for strict servers that reject unknown `chat_template_kwargs`; effort selections are then not sent for the Qwen dialects and the template runs at its own default.
- `reasoningContentField` — assistant field carrying chain-of-thought: `"reasoning_content"`, `"reasoning"`, or `"reasoning_text"`. Default: auto.
- `requiresReasoningContentForToolCalls` — assistant tool-call turns must round-trip the reasoning field (DeepSeek-R1, Kimi, OpenRouter when reasoning is on). Default: `false`.
- `allowsSyntheticReasoningContentForToolCalls` — allow a placeholder reasoning field when a prior assistant tool-call turn lacks provider reasoning content. Default: `true`; set `false` for providers that validate the exact reasoning value.
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed local OpenAI-compatible servers with strict `chat_template_kwargs` whitelists (e.g. NInfer) failing every Qwen 3.8+ turn with `400 chat_template_kwargs.reasoning_effort is not supported` after the effort routing fix: the reasoning-effort fallback now recognizes a rejection of the kwargs spelling itself, retries with the kwarg stripped while keeping the effort on the standard top-level `reasoning_effort` field (hoisting it there for the kwargs-only vLLM dialect), and remembers the shape for the rest of the session. Value-level rejections and drops now also update the `chat_template_kwargs.reasoning_effort` twin instead of leaving a stale effort for kwargs-reading renderers, and unknown-parameter 400s naming `reasoning_effort` are recognized as effort rejections.
## [17.3.8] - 2026-08-19
### Changed
@@ -1,8 +1,17 @@
import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
import type { CapturedHttpErrorResponse } from "../utils/http-inspector";
/**
* Fallback marker: the server rejected the `chat_template_kwargs.reasoning_effort`
* spelling itself (strict kwargs whitelists — Ninfer-style servers), not the
* effort value. Apply strips the kwarg and hoists the value onto the top-level
* `reasoning_effort` field when that spelling is absent.
* @internal
*/
export const STRIP_TEMPLATE_KWARG_REASONING_EFFORT = Symbol("strip-template-kwarg-reasoning-effort");
/** @internal */
export type OpenAIReasoningEffortFallback = string | null;
export type OpenAIReasoningEffortFallback = string | null | typeof STRIP_TEMPLATE_KWARG_REASONING_EFFORT;
/** @internal */
export interface OpenAIReasoningEffortFallbackState {
@@ -73,7 +82,23 @@ export function readOpenAIReasoningEffort(params: unknown): string | undefined {
if (!isRecord(params)) return undefined;
if (typeof params.reasoning_effort === "string") return params.reasoning_effort;
const reasoning = params.reasoning;
return isRecord(reasoning) && typeof reasoning.effort === "string" ? reasoning.effort : undefined;
if (isRecord(reasoning) && typeof reasoning.effort === "string") return reasoning.effort;
return readTemplateKwargReasoningEffort(params);
}
function readTemplateKwargReasoningEffort(params: Record<string, unknown>): string | undefined {
const kwargs = params.chat_template_kwargs;
return isRecord(kwargs) && typeof kwargs.reasoning_effort === "string" ? kwargs.reasoning_effort : undefined;
}
/** Remove `chat_template_kwargs.reasoning_effort`, dropping the kwargs object when it becomes empty. */
function deleteTemplateKwargReasoningEffort(kwargs: Record<string, unknown>, parent: Record<string, unknown>): void {
delete kwargs.reasoning_effort;
for (const key in kwargs) {
void key;
return;
}
delete parent.chat_template_kwargs;
}
function deleteReasoningEffort(reasoning: Record<string, unknown>, parent: Record<string, unknown>): boolean {
@@ -89,6 +114,17 @@ function deleteReasoningEffort(reasoning: Record<string, unknown>, parent: Recor
/** @internal */
export function applyOpenAIReasoningEffortFallback(params: unknown, fallback: OpenAIReasoningEffortFallback): boolean {
if (!isRecord(params)) return false;
if (fallback === STRIP_TEMPLATE_KWARG_REASONING_EFFORT) {
const kwargs = params.chat_template_kwargs;
if (!isRecord(kwargs) || typeof kwargs.reasoning_effort !== "string") return false;
const effort = kwargs.reasoning_effort;
deleteTemplateKwargReasoningEffort(kwargs, params);
// The `qwen-chat-template` dialect rides kwargs alone; keep the effort
// selection alive on the standard OpenAI field (Ninfer-style servers
// accept it, vLLM-style renderers ignore it).
if (typeof params.reasoning_effort !== "string") params.reasoning_effort = effort;
return true;
}
let changed = false;
if (typeof params.reasoning_effort === "string") {
if (fallback === null) {
@@ -107,6 +143,17 @@ export function applyOpenAIReasoningEffortFallback(params: unknown, fallback: Op
changed = true;
}
}
// Keep the Qwen template kwarg twin in lockstep — a value remap or drop
// must not leave a stale effort for kwargs-reading renderers.
const kwargs = params.chat_template_kwargs;
if (isRecord(kwargs) && typeof kwargs.reasoning_effort === "string") {
if (fallback === null) {
deleteTemplateKwargReasoningEffort(kwargs, params);
} else {
kwargs.reasoning_effort = fallback;
}
changed = true;
}
return changed;
}
@@ -165,13 +212,17 @@ function isInvalidReasoningEffortError(
if (/reasoning[_ ]content/i.test(message) && !REASONING_EFFORT_FIELD_PATTERN.test(message)) return false;
if (/invalid[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) return true;
if (
/(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|must be|expected)/i.test(
/(?:reasoning[_. ]effort|reasoning value)[^\n]*(?:invalid|unsupported|not supported|not permitted|must be|expected|unknown|unexpected|unrecognized)/i.test(
message,
)
) {
return true;
}
if (/(?:unsupported|not supported)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) {
if (
/(?:unsupported|not supported|not permitted|unknown|unexpected|unrecognized|extra)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(
message,
)
) {
return true;
}
// Gateways put the rejected value first (`level "none" not supported`), the
@@ -258,6 +309,35 @@ function nearestEnabledReasoningFallback(currentEffort: string, allowed: Set<str
return best;
}
/**
* Text that identifies a rejection of the kwargs spelling itself: the server
* names `chat_template_kwargs` together with `reasoning_effort` (Ninfer-style
* strict kwargs whitelists: `chat_template_kwargs.reasoning_effort is not
* supported`).
*/
const TEMPLATE_KWARG_EFFORT_PATTERN =
/chat_template_kwargs[^\n]{0,120}reasoning[_. ]effort|reasoning[_. ]effort[^\n]{0,120}chat_template_kwargs/i;
const FIELD_REJECTION_PATTERN =
/invalid|unsupported|not supported|not permitted|unknown|unexpected|unrecognized|rejected|extra input/i;
function resolveStripTemplateKwargFallback(
error: unknown,
captured: CapturedHttpErrorResponse | undefined,
params: unknown,
): typeof STRIP_TEMPLATE_KWARG_REASONING_EFFORT | undefined {
if (!isRecord(params)) return undefined;
const effort = readTemplateKwargReasoningEffort(params);
if (effort === undefined) return undefined;
const status = extractHttpStatusFromError(error) ?? captured?.status;
if (status !== 400 && status !== 422) return undefined;
const message = collectMessageParts(error, captured);
if (!TEMPLATE_KWARG_EFFORT_PATTERN.test(message) || !FIELD_REJECTION_PATTERN.test(message)) return undefined;
// A value-level rejection listing allowed levels wants the value remapped
// (in every spelling) by the ordinary flow, not the kwarg stripped.
if (parseAllowedReasoningValues(message, effort) !== undefined) return undefined;
return STRIP_TEMPLATE_KWARG_REASONING_EFFORT;
}
/** @internal */
export function resolveOpenAIReasoningEffortFallback(
error: unknown,
@@ -265,6 +345,8 @@ export function resolveOpenAIReasoningEffortFallback(
params: unknown,
options?: { explicitDisable?: boolean },
): OpenAIReasoningEffortFallback | undefined {
const strip = resolveStripTemplateKwargFallback(error, captured, params);
if (strip !== undefined) return strip;
const currentEffort = readOpenAIReasoningEffort(params);
if (!currentEffort || !KNOWN_REASONING_VALUE[currentEffort.toLowerCase()]) return undefined;
if (!isInvalidReasoningEffortError(error, captured, currentEffort)) return undefined;
+25 -1
View File
@@ -200,7 +200,12 @@ describe("OpenAI compat policy", () => {
expect(params.reasoning_effort).toBeUndefined();
});
function localQwenModel(id: string, provider: string, baseUrl: string): Model<"openai-completions"> {
function localQwenModel(
id: string,
provider: string,
baseUrl: string,
compat?: OpenAICompat,
): Model<"openai-completions"> {
return buildModel({
id,
name: id,
@@ -212,6 +217,7 @@ describe("OpenAI compat policy", () => {
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 262_144,
maxTokens: 32_768,
compat,
} satisfies ModelSpec<"openai-completions">);
}
@@ -252,6 +258,24 @@ describe("OpenAI compat policy", () => {
});
});
it("honors a user compat override disabling the template effort dialect", () => {
// Escape hatch for strict local servers (Ninfer-style) that reject
// unknown chat_template_kwargs: `qwenTemplateReasoningEffort: false` in
// models.yml must suppress the kwarg and revert to the pre-effort wire
// shape without disturbing thinking or preserve_thinking.
const model = localQwenModel("qwen3.8-27b", "llama.cpp", "http://127.0.0.1:8080/v1", {
qwenTemplateReasoningEffort: false,
});
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: Effort.Medium }),
);
expect(params.enable_thinking).toBe(true);
expect(params.reasoning_effort).toBeUndefined();
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
});
it("keeps pre-3.8 local Qwen on the bare enable_thinking toggle", () => {
// Qwen 3.6 templates have no reasoning_effort kwarg; leaking one would
// inject an undefined template variable for zero benefit.
@@ -133,6 +133,53 @@ function summaryReasoningErrorResponse(): Response {
);
}
/**
* Ninfer-style strict kwargs whitelist: the server rejects the
* `chat_template_kwargs.reasoning_effort` spelling itself, not the value.
*/
function templateKwargRejectionResponse(): Response {
return new Response(
JSON.stringify({
error: {
message: "chat_template_kwargs.reasoning_effort is not supported",
type: "invalid_request_error",
code: "unknown_parameter",
param: "chat_template_kwargs.reasoning_effort",
},
}),
{ status: 400, headers: { "content-type": "application/json" } },
);
}
function templateKwargValueRejectionResponse(): Response {
return new Response(
JSON.stringify({
error: {
message: "chat_template_kwargs.reasoning_effort: 'xhigh' is not supported, valid levels: low, medium, high",
type: "invalid_request_error",
param: "chat_template_kwargs.reasoning_effort",
},
}),
{ status: 400, headers: { "content-type": "application/json" } },
);
}
/** Local Qwen 3.8 model whose auto-compat routes effort onto the template kwarg. */
function createLocalQwenModel(provider: string, baseUrl: string): Model<"openai-completions"> {
return buildModel({
id: "qwen3.8-27b",
name: "Qwen3.8 27B (local)",
api: "openai-completions",
provider,
baseUrl,
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 262_144,
maxTokens: 32_768,
});
}
function parseJsonBody(init: RequestInit | undefined): Record<string, unknown> {
return JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
}
@@ -398,4 +445,103 @@ describe("OpenAI reasoning effort fallback retry", () => {
expect(result.errorStatus).toBe(400);
expect(attempts).toBe(1);
});
it("strips a rejected chat_template_kwargs.reasoning_effort, keeps the top-level twin, and remembers it", async () => {
const bodies: Record<string, unknown>[] = [];
const fetchMock: FetchImpl = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const body = parseJsonBody(init);
bodies.push(body);
return bodies.length === 1 ? templateKwargRejectionResponse() : createChatSseResponse();
},
{ preconnect: fetch.preconnect },
);
const providerSessionState = new Map<string, ProviderSessionState>();
const model = createLocalQwenModel("llama.cpp", "http://127.0.0.1:8080/v1");
const first = await streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "medium",
providerSessionState,
}).result();
expect(first.stopReason).toBe("stop");
expect(bodies).toHaveLength(2);
// The qwen dialect twin-emits; the rejected kwarg must vanish while the
// top-level field keeps the user's effort selection alive.
expect(bodies[0]!.reasoning_effort).toBe("medium");
expect(bodies[0]!.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: "medium" });
expect(bodies[1]!.reasoning_effort).toBe("medium");
expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true });
// Remembered per session: the next request pre-strips without a 400.
const second = await streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "medium",
providerSessionState,
}).result();
expect(second.stopReason).toBe("stop");
expect(bodies).toHaveLength(3);
expect(bodies[2]!.reasoning_effort).toBe("medium");
expect(bodies[2]!.chat_template_kwargs).toEqual({ preserve_thinking: true });
});
it("hoists the effort onto the top-level field when the kwargs-only dialect is rejected", async () => {
const bodies: Record<string, unknown>[] = [];
const fetchMock: FetchImpl = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const body = parseJsonBody(init);
bodies.push(body);
return bodies.length === 1 ? templateKwargRejectionResponse() : createChatSseResponse();
},
{ preconnect: fetch.preconnect },
);
const model = createLocalQwenModel("vllm", "http://127.0.0.1:8000/v1");
const result = await streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "medium",
}).result();
expect(result.stopReason).toBe("stop");
expect(bodies).toHaveLength(2);
// vLLM dialect rides kwargs alone — nothing top-level on the first try.
expect(bodies[0]!.reasoning_effort).toBeUndefined();
expect(bodies[0]!.chat_template_kwargs).toEqual({
preserve_thinking: true,
enable_thinking: true,
reasoning_effort: "medium",
});
expect(bodies[1]!.reasoning_effort).toBe("medium");
expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true, enable_thinking: true });
});
it("remaps a rejected kwargs effort value in both spellings when the error lists allowed levels", async () => {
const bodies: Record<string, unknown>[] = [];
const fetchMock: FetchImpl = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const body = parseJsonBody(init);
bodies.push(body);
return bodies.length === 1 ? templateKwargValueRejectionResponse() : createChatSseResponse();
},
{ preconnect: fetch.preconnect },
);
const model = createLocalQwenModel("llama.cpp", "http://127.0.0.1:8080/v1");
const result = await streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "xhigh",
}).result();
expect(result.stopReason).toBe("stop");
expect(bodies).toHaveLength(2);
expect(bodies[0]!.reasoning_effort).toBe("xhigh");
expect(bodies[1]!.reasoning_effort).toBe("high");
// The kwargs twin must not keep the stale rejected value.
expect(bodies[1]!.chat_template_kwargs).toEqual({ preserve_thinking: true, reasoning_effort: "high" });
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
### Changed
- `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so.
@@ -42,6 +42,7 @@ export const getModelsConfigSchemaBundle = once(() => {
"disableReasoningOnForcedToolChoice?": "boolean",
"disableReasoningOnToolChoice?": "boolean",
"thinkingFormat?": '"openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template"',
"qwenTemplateReasoningEffort?": "boolean",
"openRouterRouting?": OpenRouterRoutingSchema,
"vercelGatewayRouting?": VercelGatewayRoutingSchema,
"extraBody?": { "[string]": "unknown" },