diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md
index 5b5c5980c..7b117319f 100644
--- a/packages/ai/CHANGELOG.md
+++ b/packages/ai/CHANGELOG.md
@@ -2,6 +2,10 @@
## [Unreleased]
+### Fixed
+
+- Fixed local llama.cpp (and any local OpenAI-compatible server rendering the Qwen3.6+ chat template) re-processing the full prompt every new user message even with `replayReasoningContent` enabled (#3541 follow-up to #3528). Sending `reasoning_content` alone wasn't enough: Qwen3's chat template strips `...` from any assistant turn whose index is `<= last_query_index`, so the moment a new user message (the user's next prompt, or the auto-learn capture-at-stop nudge) lands, every prior assistant turn becomes "older" and is re-rendered without the `` block — diverging from the generation tokens still in the slot's KV cache. The chat-completions encoder now pairs `enable_thinking: true` with `preserve_thinking: true` for Qwen thinking dialects when `replayReasoningContent` is on (twin top-level + `chat_template_kwargs` emission so llama.cpp / vLLM / SGLang and Alibaba's compatible-mode wire shapes all pick it up). Qwen3.6+ then renders `...` for every assistant turn regardless of position, and the next-turn render matches the cached generation tokens. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541))
+
## [16.1.22] - 2026-06-26
### Fixed
diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts
index be8c1c6bb..d8c91d04e 100644
--- a/packages/ai/src/providers/openai-shared.ts
+++ b/packages/ai/src/providers/openai-shared.ts
@@ -585,7 +585,8 @@ export type OpenAICompletionsParams = Omit…`, diverging from the slot's existing KV cache and forcing
* full re-prefill.
*
- * The fix is the new `compat.replayReasoningContent` flag — auto-enabled for
- * the four built-in local OpenAI-compatible providers and for any provider
- * pointed at a loopback / RFC1918 baseUrl — plus a fourth branch in the
- * `openai-completions` assistant encoder that surfaces preserved thinking as
- * `reasoning_content` on every reasoning-engaged turn (not just tool-call
- * turns). This file pins the wire output across the relevant axes.
+ * Two layered fixes ship under this file:
+ * 1. `replayReasoningContent` (#3528) — auto-enabled for the four built-in
+ * local OpenAI-compatible providers and any provider pointed at a
+ * loopback / RFC1918 baseUrl, paired with a fourth branch in the
+ * `openai-completions` assistant encoder that surfaces preserved thinking
+ * as `reasoning_content` on every reasoning-engaged turn.
+ * 2. `qwenPreserveThinking` (#3541) — pairs `enable_thinking: true` with
+ * `preserve_thinking: true` (both top-level AND under
+ * `chat_template_kwargs`) so the Qwen3.6+ chat template renders
+ * `...` for older assistant turns too. Without that flag
+ * the template strips think the moment a new user message (e.g. the
+ * auto-learn nudge) shifts prior assistants past `last_query_index`,
+ * and the next-turn re-render diverges from the slot's cached
+ * generation tokens — the exact symptom logged in #3541.
+ *
+ * This file pins the wire output across the relevant axes.
*/
import { describe, expect, it } from "bun:test";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
+import {
+ applyChatCompletionsReasoningParams,
+ type OpenAICompletionsParams,
+} from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { AssistantMessage, Message, Model, ModelSpec, ThinkingContent, UserMessage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -386,4 +400,112 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
expect(found?.reasoning_content).toBeUndefined();
expect(found?.reasoning).toBeUndefined();
});
+
+ it("auto-enables qwenPreserveThinking for llama.cpp + Qwen", () => {
+ // Pair to `replayReasoningContent`: without it the Qwen3.6+ template
+ // strips `...` from older assistant turns the moment a
+ // new user message (or auto-learn nudge) shifts them past
+ // `last_query_index`, and the re-render diverges from the slot's KV
+ // cache state.
+ const compat = llamaCppQwenModel().compat;
+ expect(compat.qwenPreserveThinking).toBe(true);
+ });
+
+ it("auto-enables qwenPreserveThinking for the other built-in local providers + Qwen", () => {
+ const lmStudio = llamaCppQwenModel({ provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }).compat;
+ const vllm = llamaCppQwenModel({ provider: "vllm", baseUrl: "http://127.0.0.1:8000/v1" }).compat;
+ const ollama = llamaCppQwenModel({ provider: "ollama", baseUrl: "http://localhost:11434/v1" }).compat;
+ expect(lmStudio.qwenPreserveThinking).toBe(true);
+ expect(vllm.qwenPreserveThinking).toBe(true);
+ expect(ollama.qwenPreserveThinking).toBe(true);
+ });
+
+ it("auto-enables qwenPreserveThinking for custom providers on loopback baseUrls + Qwen", () => {
+ const loopback = llamaCppQwenModel({ provider: "custom", baseUrl: "http://localhost:9000/v1" }).compat;
+ const rfc1918 = llamaCppQwenModel({ provider: "custom", baseUrl: "http://10.0.0.42:8080/v1" }).compat;
+ expect(loopback.qwenPreserveThinking).toBe(true);
+ expect(rfc1918.qwenPreserveThinking).toBe(true);
+ });
+
+ it("leaves qwenPreserveThinking off for non-Qwen models on local llama.cpp", () => {
+ // Non-Qwen templates ignore the param either way, but auto-detection
+ // gates on the Qwen thinking dialect so the wire body stays minimal.
+ const deepseek = llamaCppQwenModel({ id: "deepseek-r1-32b", name: "DeepSeek R1 32B" }).compat;
+ expect(deepseek.qwenPreserveThinking).toBe(false);
+ });
+
+ it("leaves qwenPreserveThinking off for cloud Qwen hosts", () => {
+ // Alibaba's Dashscope and Qwen Portal own the slot lifecycle on the
+ // cloud side; OMP isn't responsible for KV-cache invalidation there,
+ // and `preserve_thinking` is opt-in per the Alibaba docs. Stay
+ // minimal on the wire unless the user opts in via `compat`.
+ const dashscope = llamaCppQwenModel({
+ provider: "alibaba",
+ baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
+ }).compat;
+ expect(dashscope.qwenPreserveThinking).toBe(false);
+ });
+
+ it("emits preserve_thinking on the wire for local Qwen + thinking", () => {
+ // End-to-end pin for the user's reported setup (#3541):
+ // `enable_thinking: true` + `preserve_thinking: true` (twin top-level
+ // + chat_template_kwargs) must both ride the body so the chat template
+ // preserves `...` for older assistants. The twin
+ // emission covers llama.cpp / vLLM / SGLang / Alibaba shapes without
+ // per-host sniffing.
+ const model = llamaCppQwenModel();
+ const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
+ applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
+ expect(params.enable_thinking).toBe(true);
+ expect(params.preserve_thinking).toBe(true);
+ expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
+ });
+
+ it("does NOT emit preserve_thinking for cloud Qwen + thinking", () => {
+ const model = llamaCppQwenModel({
+ provider: "alibaba",
+ baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
+ });
+ const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
+ applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
+ expect(params.enable_thinking).toBe(true);
+ expect(params.preserve_thinking).toBeUndefined();
+ // `chat_template_kwargs` stays unset — Alibaba's qwen dialect rides
+ // only the top-level `enable_thinking`.
+ expect(params.chat_template_kwargs).toBeUndefined();
+ });
+
+ it("does NOT emit preserve_thinking when reasoning is disabled on local Qwen", () => {
+ const model = llamaCppQwenModel();
+ const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
+ applyChatCompletionsReasoningParams(params, model, model.compat, { disableReasoning: true });
+ // `enable_thinking: false` is the Qwen "disable" encoding; the
+ // preserve knob is moot on a non-thinking turn and must stay off so
+ // stale `` markup isn't reintroduced into the prompt.
+ expect(params.enable_thinking).toBe(false);
+ expect(params.preserve_thinking).toBeUndefined();
+ });
+
+ it("honors an explicit qwenPreserveThinking override on cloud Qwen", () => {
+ // Escape hatch for power users who run a cloud-fronted llama.cpp /
+ // vLLM and know the template benefits from the replay.
+ const model = llamaCppQwenModel({
+ provider: "alibaba",
+ baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
+ compat: { qwenPreserveThinking: true },
+ });
+ const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
+ applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
+ expect(params.preserve_thinking).toBe(true);
+ expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true });
+ });
+
+ it("honors an explicit qwenPreserveThinking opt-out on local Qwen", () => {
+ const model = llamaCppQwenModel({ compat: { qwenPreserveThinking: false } });
+ expect(model.compat.qwenPreserveThinking).toBe(false);
+ const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true };
+ applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" });
+ expect(params.enable_thinking).toBe(true);
+ expect(params.preserve_thinking).toBeUndefined();
+ });
});
diff --git a/packages/ai/test/issue-967-vision-guard.test.ts b/packages/ai/test/issue-967-vision-guard.test.ts
index 6e7a55642..f060b916b 100644
--- a/packages/ai/test/issue-967-vision-guard.test.ts
+++ b/packages/ai/test/issue-967-vision-guard.test.ts
@@ -51,6 +51,7 @@ const compat: ResolvedOpenAICompat = {
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
+ qwenPreserveThinking: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts
index 31f65e15c..b0206b630 100644
--- a/packages/ai/test/openai-completions-compat.test.ts
+++ b/packages/ai/test/openai-completions-compat.test.ts
@@ -185,6 +185,7 @@ describe("openai-completions compatibility", () => {
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
+ qwenPreserveThinking: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts
index 5af3f973a..5b6d76908 100644
--- a/packages/ai/test/openai-completions-tool-result-images.test.ts
+++ b/packages/ai/test/openai-completions-tool-result-images.test.ts
@@ -40,6 +40,7 @@ const compat: ResolvedOpenAICompat = {
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
replayReasoningContent: false,
+ qwenPreserveThinking: false,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
vercelGatewayRouting: {},
diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md
index f8e6356d8..072669fa5 100644
--- a/packages/catalog/CHANGELOG.md
+++ b/packages/catalog/CHANGELOG.md
@@ -2,6 +2,10 @@
## [Unreleased]
+### Added
+
+- Added `OpenAICompat.qwenPreserveThinking` — auto-enabled when the resolved `thinkingFormat` is `"qwen"` or `"qwen-chat-template"` AND `replayReasoningContent` is on (i.e. the four built-in local OpenAI-compatible providers, or a custom provider pointed at a loopback / RFC1918 / `*.local` baseUrl). Pairs with the chat-completions encoder change so the request body carries `preserve_thinking: true` (twin top-level + `chat_template_kwargs` emission), keeping Qwen3.6+ from stripping `...` off older assistant turns and breaking the local slot's KV cache between user messages. Non-Qwen chat templates ignore the parameter, so the flag stays a no-op outside the Qwen path; users on a cloud Qwen host (Alibaba Dashscope / Qwen Portal) can opt in with `compat.qwenPreserveThinking: true`. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541))
+
## [16.1.22] - 2026-06-26
### Added
diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts
index 72feb366f..b6086c694 100644
--- a/packages/catalog/src/compat/openai.ts
+++ b/packages/catalog/src/compat/openai.ts
@@ -468,6 +468,20 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
replayReasoningContent:
!PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) &&
(LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)),
+ // `preserve_thinking: true` makes the Qwen3.6+ chat template render
+ // `...` for older assistant turns too, instead of
+ // stripping it the moment a new user message moves them past
+ // `last_query_index`. Without it, the slot's KV cache (which holds the
+ // raw `X` tokens emitted during generation) diverges
+ // from the next-turn render and llama.cpp falls back to full prompt
+ // re-processing — the exact symptom reported in #3541. Auto-enabled
+ // for Qwen thinking dialects on local llama.cpp-style backends (paired
+ // with `replayReasoningContent` above). Non-Qwen templates ignore the
+ // parameter, so the flag stays a no-op outside the Qwen path.
+ qwenPreserveThinking:
+ (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") &&
+ !PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) &&
+ (LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)),
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined,
openRouterRouting: undefined,
@@ -589,6 +603,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
// not via a top-level `reasoning_content` field — this flag is
// chat-completions-only.
replayReasoningContent: false,
+ // Responses-only; the Qwen `preserve_thinking` template knob lives on
+ // the chat-completions wire shape, never on Responses.
+ qwenPreserveThinking: false,
requiresThinkingAsText: false,
requiresMistralToolIds: false,
requiresToolResultName: false,
diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts
index b4d7b46de..45fd362bc 100644
--- a/packages/catalog/src/types.ts
+++ b/packages/catalog/src/types.ts
@@ -236,6 +236,28 @@ export interface OpenAICompat {
* models).
*/
replayReasoningContent?: boolean;
+ /**
+ * Send `preserve_thinking: true` so the Qwen3.6+ chat template renders
+ * `...` markup for EVERY assistant turn (not just turns
+ * after the last user message). Without it, the template strips the think
+ * block from older assistant turns:
+ *
+ * ```jinja
+ * {%- if (preserve_thinking is defined and preserve_thinking is true)
+ * or (loop.index0 > ns.last_query_index) %}
+ * <|im_start|>assistant\n\n{rc}\n\n\n{content}
+ * {%- else %}
+ * <|im_start|>assistant\n{content}
+ * ```
+ *
+ * The cache from the original generation has `...` tokens,
+ * so once a new user message arrives the prior assistant turns become
+ * "older" and the stripped re-render diverges — full prompt re-processing
+ * on SWA models (#3541). Default: auto-detected (Qwen thinking format on
+ * a local llama.cpp-style backend, paired with `replayReasoningContent`).
+ * Non-Qwen templates ignore the flag, so the auto-detection is safe.
+ */
+ qwenPreserveThinking?: boolean;
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
requiresAssistantContentForToolCalls?: boolean;
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
@@ -448,6 +470,7 @@ export interface ResolvedOpenAISharedCompat {
requiresReasoningContentForAllAssistantTurns: boolean;
allowsSyntheticReasoningContentForToolCalls: boolean;
replayReasoningContent: boolean;
+ qwenPreserveThinking: boolean;
requiresThinkingAsText: boolean;
requiresMistralToolIds: boolean;
requiresToolResultName: boolean;
@@ -498,6 +521,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "requiresReasoningContentForAllAssistantTurns"
| "allowsSyntheticReasoningContentForToolCalls"
| "replayReasoningContent"
+ | "qwenPreserveThinking"
| "requiresThinkingAsText"
| "requiresMistralToolIds"
| "requiresToolResultName"