fix(ai): emit Qwen preserve_thinking so local-server cache survives new user messages
The Qwen3 / Qwen3.6 chat template strips <think>...</think> from every assistant turn whose loop.index0 <= ns.last_query_index, so the moment a new user message (the user's real next prompt OR the auto-learn capture-at-stop nudge) lands, every prior assistant turn becomes 'older' and is re-rendered without its <think> block — diverging from the generation tokens still in the local slot's KV cache and forcing full prompt re-processing on SWA models. Sending reasoning_content alone (the #3528 fix) does not help: the template's older branch renders only `content`, never the reasoning_content field. The official Qwen3.6 fix is `preserve_thinking: true`, which makes the template render <think>\n{reasoning_content}\n</think>\n\n{content} for every assistant turn regardless of position. - packages/catalog/src/compat/openai.ts: new qwenPreserveThinking shared compat flag, auto-enabled when the resolved thinkingFormat is `qwen` or `qwen-chat-template` AND replayReasoningContent is on (the four local provider ids plus loopback / RFC1918 / *.local baseUrls). Responses-API builder pins it false — it's a chat-template knob, irrelevant on the Responses surface. - packages/ai/src/providers/openai-shared.ts: chat-completions encoder emits preserve_thinking: true alongside enable_thinking: true in both Qwen disable-mode branches (twin top-level + chat_template_kwargs emission so llama.cpp / vLLM / SGLang and Alibaba's compatible-mode wire shapes all pick it up). Stays off when thinking is disabled. - packages/ai/test/issue-3528-repro.test.ts: nine new pins covering the auto-detection matrix, the wire emission on local Qwen + thinking, the cloud-Qwen / reasoning-disabled negative cases, and the explicit-override escape hatches in both directions. - Backfilled qwenPreserveThinking: false on three hand-rolled ResolvedOpenAICompat fixtures so the required field stays satisfied. - AI + catalog changelog entries under ## [Unreleased]. Verification: bun --cwd packages/ai test ./test/issue-3528-repro.test.ts ./test/openai-completions-compat.test.ts ./test/openai-completions-tool-result-images.test.ts ./test/issue-967-vision-guard.test.ts → 87 pass bun --cwd packages/ai test ./test/issue-3434-repro.test.ts ./test/deepseek-reasoning-content.test.ts ./test/ollama-thinking-disable.test.ts ./test/openai-compat-policy.test.ts → 37 pass bun --cwd packages/catalog test → 325/325 pass bun --cwd packages/ai check:types && bun --cwd packages/catalog check:types → clean Fixes #3541
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed local llama.cpp (and any local OpenAI-compatible server rendering the Qwen3.6+ chat template) re-processing the full prompt every new user message even with `replayReasoningContent` enabled (#3541 follow-up to #3528). Sending `reasoning_content` alone wasn't enough: Qwen3's chat template strips `<think>...</think>` from any assistant turn whose index is `<= last_query_index`, so the moment a new user message (the user's next prompt, or the auto-learn capture-at-stop nudge) lands, every prior assistant turn becomes "older" and is re-rendered without the `<think>` block — diverging from the generation tokens still in the slot's KV cache. The chat-completions encoder now pairs `enable_thinking: true` with `preserve_thinking: true` for Qwen thinking dialects when `replayReasoningContent` is on (twin top-level + `chat_template_kwargs` emission so llama.cpp / vLLM / SGLang and Alibaba's compatible-mode wire shapes all pick it up). Qwen3.6+ then renders `<think>...</think>` for every assistant turn regardless of position, and the next-turn render matches the cached generation tokens. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541))
|
||||
|
||||
## [16.1.22] - 2026-06-26
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -585,7 +585,8 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
||||
repetition_penalty?: number;
|
||||
thinking?: { type: "enabled" | "disabled"; keep?: "all" };
|
||||
enable_thinking?: boolean;
|
||||
chat_template_kwargs?: { enable_thinking: boolean };
|
||||
preserve_thinking?: boolean;
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
reasoning_effort?: string | null;
|
||||
service_tier?: ResolvedServiceTier;
|
||||
@@ -830,9 +831,26 @@ export function applyChatCompletionsCompatPolicy(params: OpenAICompletionsParams
|
||||
break;
|
||||
case "qwen-enable-thinking-false":
|
||||
params.enable_thinking = true;
|
||||
if (policy.compat.qwenPreserveThinking) {
|
||||
// Twin top-level + chat_template_kwargs emission: llama.cpp's
|
||||
// chat_template hook binds the kwargs into the jinja namespace,
|
||||
// vLLM/SGLang only see fields under chat_template_kwargs, and
|
||||
// Alibaba Cloud Model Studio's compatible-mode reads the
|
||||
// top-level field instead. Setting both covers every host that
|
||||
// renders the Qwen3.6+ template without sniffing per-host
|
||||
// shapes — the parameter is silently ignored elsewhere.
|
||||
params.preserve_thinking = true;
|
||||
params.chat_template_kwargs = {
|
||||
...params.chat_template_kwargs,
|
||||
preserve_thinking: true,
|
||||
};
|
||||
}
|
||||
break;
|
||||
case "qwen-template-false":
|
||||
params.chat_template_kwargs = { enable_thinking: true };
|
||||
if (policy.compat.qwenPreserveThinking) {
|
||||
params.chat_template_kwargs.preserve_thinking = true;
|
||||
}
|
||||
break;
|
||||
case "openrouter-enabled-false":
|
||||
if (reasoning.wireEffort !== undefined) {
|
||||
|
||||
@@ -22,15 +22,29 @@
|
||||
* `<think>…</think>`, diverging from the slot's existing KV cache and forcing
|
||||
* full re-prefill.
|
||||
*
|
||||
* The fix is the new `compat.replayReasoningContent` flag — auto-enabled for
|
||||
* the four built-in local OpenAI-compatible providers and for any provider
|
||||
* pointed at a loopback / RFC1918 baseUrl — plus a fourth branch in the
|
||||
* `openai-completions` assistant encoder that surfaces preserved thinking as
|
||||
* `reasoning_content` on every reasoning-engaged turn (not just tool-call
|
||||
* turns). This file pins the wire output across the relevant axes.
|
||||
* Two layered fixes ship under this file:
|
||||
* 1. `replayReasoningContent` (#3528) — auto-enabled for the four built-in
|
||||
* local OpenAI-compatible providers and any provider pointed at a
|
||||
* loopback / RFC1918 baseUrl, paired with a fourth branch in the
|
||||
* `openai-completions` assistant encoder that surfaces preserved thinking
|
||||
* as `reasoning_content` on every reasoning-engaged turn.
|
||||
* 2. `qwenPreserveThinking` (#3541) — pairs `enable_thinking: true` with
|
||||
* `preserve_thinking: true` (both top-level AND under
|
||||
* `chat_template_kwargs`) so the Qwen3.6+ chat template renders
|
||||
* `<think>...</think>` for older assistant turns too. Without that flag
|
||||
* the template strips think the moment a new user message (e.g. the
|
||||
* auto-learn nudge) shifts prior assistants past `last_query_index`,
|
||||
* and the next-turn re-render diverges from the slot's cached
|
||||
* generation tokens — the exact symptom logged in #3541.
|
||||
*
|
||||
* This file pins the wire output across the relevant axes.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import {
|
||||
applyChatCompletionsReasoningParams,
|
||||
type OpenAICompletionsParams,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Message, Model, ModelSpec, ThinkingContent, UserMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
@@ -386,4 +400,112 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
|
||||
expect(found?.reasoning_content).toBeUndefined();
|
||||
expect(found?.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
it("auto-enables qwenPreserveThinking for llama.cpp + Qwen", () => {
|
||||
// Pair to `replayReasoningContent`: without it the Qwen3.6+ template
|
||||
// strips `<think>...</think>` from older assistant turns the moment a
|
||||
// new user message (or auto-learn nudge) shifts them past
|
||||
// `last_query_index`, and the re-render diverges from the slot's KV
|
||||
// cache state.
|
||||
const compat = llamaCppQwenModel().compat;
|
||||
expect(compat.qwenPreserveThinking).toBe(true);
|
||||
});
|
||||
|
||||
it("auto-enables qwenPreserveThinking for the other built-in local providers + Qwen", () => {
|
||||
const lmStudio = llamaCppQwenModel({ provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }).compat;
|
||||
const vllm = llamaCppQwenModel({ provider: "vllm", baseUrl: "http://127.0.0.1:8000/v1" }).compat;
|
||||
const ollama = llamaCppQwenModel({ provider: "ollama", baseUrl: "http://localhost:11434/v1" }).compat;
|
||||
expect(lmStudio.qwenPreserveThinking).toBe(true);
|
||||
expect(vllm.qwenPreserveThinking).toBe(true);
|
||||
expect(ollama.qwenPreserveThinking).toBe(true);
|
||||
});
|
||||
|
||||
it("auto-enables qwenPreserveThinking for custom providers on loopback baseUrls + Qwen", () => {
|
||||
const loopback = llamaCppQwenModel({ provider: "custom", baseUrl: "http://localhost:9000/v1" }).compat;
|
||||
const rfc1918 = llamaCppQwenModel({ provider: "custom", baseUrl: "http://10.0.0.42:8080/v1" }).compat;
|
||||
expect(loopback.qwenPreserveThinking).toBe(true);
|
||||
expect(rfc1918.qwenPreserveThinking).toBe(true);
|
||||
});
|
||||
|
||||
it("leaves qwenPreserveThinking off for non-Qwen models on local llama.cpp", () => {
|
||||
// Non-Qwen templates ignore the param either way, but auto-detection
|
||||
// gates on the Qwen thinking dialect so the wire body stays minimal.
|
||||
const deepseek = llamaCppQwenModel({ id: "deepseek-r1-32b", name: "DeepSeek R1 32B" }).compat;
|
||||
expect(deepseek.qwenPreserveThinking).toBe(false);
|
||||
});
|
||||
|
||||
it("leaves qwenPreserveThinking off for cloud Qwen hosts", () => {
|
||||
// Alibaba's Dashscope and Qwen Portal own the slot lifecycle on the
|
||||
// cloud side; OMP isn't responsible for KV-cache invalidation there,
|
||||
// and `preserve_thinking` is opt-in per the Alibaba docs. Stay
|
||||
// minimal on the wire unless the user opts in via `compat`.
|
||||
const dashscope = llamaCppQwenModel({
|
||||
provider: "alibaba",
|
||||
baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
||||
}).compat;
|
||||
expect(dashscope.qwenPreserveThinking).toBe(false);
|
||||
});
|
||||
|
||||
it("emits preserve_thinking on the wire for local Qwen + thinking", () => {
|
||||
// End-to-end pin for the user's reported setup (#3541):
|
||||
// `enable_thinking: true` + `preserve_thinking: true` (twin top-level
|
||||
// + chat_template_kwargs) must both ride the body so the chat template
|
||||
// preserves `<think>...</think>` for older assistants. The twin
|
||||
// emission covers llama.cpp / vLLM / SGLang / Alibaba shapes without
|
||||
// per-host sniffing.
|
||||
const model = llamaCppQwenModel();
|
||||
const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
|
||||
applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.preserve_thinking).toBe(true);
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
|
||||
});
|
||||
|
||||
it("does NOT emit preserve_thinking for cloud Qwen + thinking", () => {
|
||||
const model = llamaCppQwenModel({
|
||||
provider: "alibaba",
|
||||
baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
||||
});
|
||||
const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
|
||||
applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.preserve_thinking).toBeUndefined();
|
||||
// `chat_template_kwargs` stays unset — Alibaba's qwen dialect rides
|
||||
// only the top-level `enable_thinking`.
|
||||
expect(params.chat_template_kwargs).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does NOT emit preserve_thinking when reasoning is disabled on local Qwen", () => {
|
||||
const model = llamaCppQwenModel();
|
||||
const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
|
||||
applyChatCompletionsReasoningParams(params, model, model.compat, { disableReasoning: true });
|
||||
// `enable_thinking: false` is the Qwen "disable" encoding; the
|
||||
// preserve knob is moot on a non-thinking turn and must stay off so
|
||||
// stale `<think>` markup isn't reintroduced into the prompt.
|
||||
expect(params.enable_thinking).toBe(false);
|
||||
expect(params.preserve_thinking).toBeUndefined();
|
||||
});
|
||||
|
||||
it("honors an explicit qwenPreserveThinking override on cloud Qwen", () => {
|
||||
// Escape hatch for power users who run a cloud-fronted llama.cpp /
|
||||
// vLLM and know the template benefits from the replay.
|
||||
const model = llamaCppQwenModel({
|
||||
provider: "alibaba",
|
||||
baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
||||
compat: { qwenPreserveThinking: true },
|
||||
});
|
||||
const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
|
||||
applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
|
||||
expect(params.preserve_thinking).toBe(true);
|
||||
expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
|
||||
});
|
||||
|
||||
it("honors an explicit qwenPreserveThinking opt-out on local Qwen", () => {
|
||||
const model = llamaCppQwenModel({ compat: { qwenPreserveThinking: false } });
|
||||
expect(model.compat.qwenPreserveThinking).toBe(false);
|
||||
const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
|
||||
applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
|
||||
expect(params.enable_thinking).toBe(true);
|
||||
expect(params.preserve_thinking).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -51,6 +51,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -185,6 +185,7 @@ describe("openai-completions compatibility", () => {
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -40,6 +40,7 @@ const compat: ResolvedOpenAICompat = {
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
replayReasoningContent: false,
|
||||
qwenPreserveThinking: false,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `OpenAICompat.qwenPreserveThinking` — auto-enabled when the resolved `thinkingFormat` is `"qwen"` or `"qwen-chat-template"` AND `replayReasoningContent` is on (i.e. the four built-in local OpenAI-compatible providers, or a custom provider pointed at a loopback / RFC1918 / `*.local` baseUrl). Pairs with the chat-completions encoder change so the request body carries `preserve_thinking: true` (twin top-level + `chat_template_kwargs` emission), keeping Qwen3.6+ from stripping `<think>...</think>` off older assistant turns and breaking the local slot's KV cache between user messages. Non-Qwen chat templates ignore the parameter, so the flag stays a no-op outside the Qwen path; users on a cloud Qwen host (Alibaba Dashscope / Qwen Portal) can opt in with `compat.qwenPreserveThinking: true`. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541))
|
||||
|
||||
## [16.1.22] - 2026-06-26
|
||||
|
||||
### Added
|
||||
|
||||
@@ -468,6 +468,20 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
replayReasoningContent:
|
||||
!PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) &&
|
||||
(LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)),
|
||||
// `preserve_thinking: true` makes the Qwen3.6+ chat template render
|
||||
// `<think>...</think>` for older assistant turns too, instead of
|
||||
// stripping it the moment a new user message moves them past
|
||||
// `last_query_index`. Without it, the slot's KV cache (which holds the
|
||||
// raw `<think>X</think>` tokens emitted during generation) diverges
|
||||
// from the next-turn render and llama.cpp falls back to full prompt
|
||||
// re-processing — the exact symptom reported in #3541. Auto-enabled
|
||||
// for Qwen thinking dialects on local llama.cpp-style backends (paired
|
||||
// with `replayReasoningContent` above). Non-Qwen templates ignore the
|
||||
// parameter, so the flag stays a no-op outside the Qwen path.
|
||||
qwenPreserveThinking:
|
||||
(thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") &&
|
||||
!PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) &&
|
||||
(LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)),
|
||||
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
|
||||
cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined,
|
||||
openRouterRouting: undefined,
|
||||
@@ -589,6 +603,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
// not via a top-level `reasoning_content` field — this flag is
|
||||
// chat-completions-only.
|
||||
replayReasoningContent: false,
|
||||
// Responses-only; the Qwen `preserve_thinking` template knob lives on
|
||||
// the chat-completions wire shape, never on Responses.
|
||||
qwenPreserveThinking: false,
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
requiresToolResultName: false,
|
||||
|
||||
@@ -236,6 +236,28 @@ export interface OpenAICompat {
|
||||
* models).
|
||||
*/
|
||||
replayReasoningContent?: boolean;
|
||||
/**
|
||||
* Send `preserve_thinking: true` so the Qwen3.6+ chat template renders
|
||||
* `<think>...</think>` markup for EVERY assistant turn (not just turns
|
||||
* after the last user message). Without it, the template strips the think
|
||||
* block from older assistant turns:
|
||||
*
|
||||
* ```jinja
|
||||
* {%- if (preserve_thinking is defined and preserve_thinking is true)
|
||||
* or (loop.index0 > ns.last_query_index) %}
|
||||
* <|im_start|>assistant\n<think>\n{rc}\n</think>\n\n{content}
|
||||
* {%- else %}
|
||||
* <|im_start|>assistant\n{content}
|
||||
* ```
|
||||
*
|
||||
* The cache from the original generation has `<think>...</think>` tokens,
|
||||
* so once a new user message arrives the prior assistant turns become
|
||||
* "older" and the stripped re-render diverges — full prompt re-processing
|
||||
* on SWA models (#3541). Default: auto-detected (Qwen thinking format on
|
||||
* a local llama.cpp-style backend, paired with `replayReasoningContent`).
|
||||
* Non-Qwen templates ignore the flag, so the auto-detection is safe.
|
||||
*/
|
||||
qwenPreserveThinking?: boolean;
|
||||
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
|
||||
requiresAssistantContentForToolCalls?: boolean;
|
||||
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
|
||||
@@ -448,6 +470,7 @@ export interface ResolvedOpenAISharedCompat {
|
||||
requiresReasoningContentForAllAssistantTurns: boolean;
|
||||
allowsSyntheticReasoningContentForToolCalls: boolean;
|
||||
replayReasoningContent: boolean;
|
||||
qwenPreserveThinking: boolean;
|
||||
requiresThinkingAsText: boolean;
|
||||
requiresMistralToolIds: boolean;
|
||||
requiresToolResultName: boolean;
|
||||
@@ -498,6 +521,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
||||
| "requiresReasoningContentForAllAssistantTurns"
|
||||
| "allowsSyntheticReasoningContentForToolCalls"
|
||||
| "replayReasoningContent"
|
||||
| "qwenPreserveThinking"
|
||||
| "requiresThinkingAsText"
|
||||
| "requiresMistralToolIds"
|
||||
| "requiresToolResultName"
|
||||
|
||||
Reference in New Issue
Block a user