diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 5b5c5980c..7b117319f 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed local llama.cpp (and any local OpenAI-compatible server rendering the Qwen3.6+ chat template) re-processing the full prompt every new user message even with `replayReasoningContent` enabled (#3541 follow-up to #3528). Sending `reasoning_content` alone wasn't enough: Qwen3's chat template strips `...` from any assistant turn whose index is `<= last_query_index`, so the moment a new user message (the user's next prompt, or the auto-learn capture-at-stop nudge) lands, every prior assistant turn becomes "older" and is re-rendered without the `` block — diverging from the generation tokens still in the slot's KV cache. The chat-completions encoder now pairs `enable_thinking: true` with `preserve_thinking: true` for Qwen thinking dialects when `replayReasoningContent` is on (twin top-level + `chat_template_kwargs` emission so llama.cpp / vLLM / SGLang and Alibaba's compatible-mode wire shapes all pick it up). Qwen3.6+ then renders `...` for every assistant turn regardless of position, and the next-turn render matches the cached generation tokens. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541)) + ## [16.1.22] - 2026-06-26 ### Fixed diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index be8c1c6bb..d8c91d04e 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -585,7 +585,8 @@ export type OpenAICompletionsParams = Omit…`, diverging from the slot's existing KV cache and forcing * full re-prefill. * - * The fix is the new `compat.replayReasoningContent` flag — auto-enabled for - * the four built-in local OpenAI-compatible providers and for any provider - * pointed at a loopback / RFC1918 baseUrl — plus a fourth branch in the - * `openai-completions` assistant encoder that surfaces preserved thinking as - * `reasoning_content` on every reasoning-engaged turn (not just tool-call - * turns). This file pins the wire output across the relevant axes. + * Two layered fixes ship under this file: + * 1. `replayReasoningContent` (#3528) — auto-enabled for the four built-in + * local OpenAI-compatible providers and any provider pointed at a + * loopback / RFC1918 baseUrl, paired with a fourth branch in the + * `openai-completions` assistant encoder that surfaces preserved thinking + * as `reasoning_content` on every reasoning-engaged turn. + * 2. `qwenPreserveThinking` (#3541) — pairs `enable_thinking: true` with + * `preserve_thinking: true` (both top-level AND under + * `chat_template_kwargs`) so the Qwen3.6+ chat template renders + * `...` for older assistant turns too. Without that flag + * the template strips think the moment a new user message (e.g. the + * auto-learn nudge) shifts prior assistants past `last_query_index`, + * and the next-turn re-render diverges from the slot's cached + * generation tokens — the exact symptom logged in #3541. + * + * This file pins the wire output across the relevant axes. */ import { describe, expect, it } from "bun:test"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { + applyChatCompletionsReasoningParams, + type OpenAICompletionsParams, +} from "@oh-my-pi/pi-ai/providers/openai-shared"; import type { AssistantMessage, Message, Model, ModelSpec, ThinkingContent, UserMessage } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -386,4 +400,112 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => { expect(found?.reasoning_content).toBeUndefined(); expect(found?.reasoning).toBeUndefined(); }); + + it("auto-enables qwenPreserveThinking for llama.cpp + Qwen", () => { + // Pair to `replayReasoningContent`: without it the Qwen3.6+ template + // strips `...` from older assistant turns the moment a + // new user message (or auto-learn nudge) shifts them past + // `last_query_index`, and the re-render diverges from the slot's KV + // cache state. + const compat = llamaCppQwenModel().compat; + expect(compat.qwenPreserveThinking).toBe(true); + }); + + it("auto-enables qwenPreserveThinking for the other built-in local providers + Qwen", () => { + const lmStudio = llamaCppQwenModel({ provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }).compat; + const vllm = llamaCppQwenModel({ provider: "vllm", baseUrl: "http://127.0.0.1:8000/v1" }).compat; + const ollama = llamaCppQwenModel({ provider: "ollama", baseUrl: "http://localhost:11434/v1" }).compat; + expect(lmStudio.qwenPreserveThinking).toBe(true); + expect(vllm.qwenPreserveThinking).toBe(true); + expect(ollama.qwenPreserveThinking).toBe(true); + }); + + it("auto-enables qwenPreserveThinking for custom providers on loopback baseUrls + Qwen", () => { + const loopback = llamaCppQwenModel({ provider: "custom", baseUrl: "http://localhost:9000/v1" }).compat; + const rfc1918 = llamaCppQwenModel({ provider: "custom", baseUrl: "http://10.0.0.42:8080/v1" }).compat; + expect(loopback.qwenPreserveThinking).toBe(true); + expect(rfc1918.qwenPreserveThinking).toBe(true); + }); + + it("leaves qwenPreserveThinking off for non-Qwen models on local llama.cpp", () => { + // Non-Qwen templates ignore the param either way, but auto-detection + // gates on the Qwen thinking dialect so the wire body stays minimal. + const deepseek = llamaCppQwenModel({ id: "deepseek-r1-32b", name: "DeepSeek R1 32B" }).compat; + expect(deepseek.qwenPreserveThinking).toBe(false); + }); + + it("leaves qwenPreserveThinking off for cloud Qwen hosts", () => { + // Alibaba's Dashscope and Qwen Portal own the slot lifecycle on the + // cloud side; OMP isn't responsible for KV-cache invalidation there, + // and `preserve_thinking` is opt-in per the Alibaba docs. Stay + // minimal on the wire unless the user opts in via `compat`. + const dashscope = llamaCppQwenModel({ + provider: "alibaba", + baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", + }).compat; + expect(dashscope.qwenPreserveThinking).toBe(false); + }); + + it("emits preserve_thinking on the wire for local Qwen + thinking", () => { + // End-to-end pin for the user's reported setup (#3541): + // `enable_thinking: true` + `preserve_thinking: true` (twin top-level + // + chat_template_kwargs) must both ride the body so the chat template + // preserves `...` for older assistants. The twin + // emission covers llama.cpp / vLLM / SGLang / Alibaba shapes without + // per-host sniffing. + const model = llamaCppQwenModel(); + const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true }; + applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" }); + expect(params.enable_thinking).toBe(true); + expect(params.preserve_thinking).toBe(true); + expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true }); + }); + + it("does NOT emit preserve_thinking for cloud Qwen + thinking", () => { + const model = llamaCppQwenModel({ + provider: "alibaba", + baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", + }); + const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true }; + applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" }); + expect(params.enable_thinking).toBe(true); + expect(params.preserve_thinking).toBeUndefined(); + // `chat_template_kwargs` stays unset — Alibaba's qwen dialect rides + // only the top-level `enable_thinking`. + expect(params.chat_template_kwargs).toBeUndefined(); + }); + + it("does NOT emit preserve_thinking when reasoning is disabled on local Qwen", () => { + const model = llamaCppQwenModel(); + const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true }; + applyChatCompletionsReasoningParams(params, model, model.compat, { disableReasoning: true }); + // `enable_thinking: false` is the Qwen "disable" encoding; the + // preserve knob is moot on a non-thinking turn and must stay off so + // stale `` markup isn't reintroduced into the prompt. + expect(params.enable_thinking).toBe(false); + expect(params.preserve_thinking).toBeUndefined(); + }); + + it("honors an explicit qwenPreserveThinking override on cloud Qwen", () => { + // Escape hatch for power users who run a cloud-fronted llama.cpp / + // vLLM and know the template benefits from the replay. + const model = llamaCppQwenModel({ + provider: "alibaba", + baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", + compat: { qwenPreserveThinking: true }, + }); + const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true }; + applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" }); + expect(params.preserve_thinking).toBe(true); + expect(params.chat_template_kwargs).toEqual({ preserve_thinking: true }); + }); + + it("honors an explicit qwenPreserveThinking opt-out on local Qwen", () => { + const model = llamaCppQwenModel({ compat: { qwenPreserveThinking: false } }); + expect(model.compat.qwenPreserveThinking).toBe(false); + const params: OpenAICompletionsParams = { model: model.id, messages: [], stream: true }; + applyChatCompletionsReasoningParams(params, model, model.compat, { reasoning: "medium" }); + expect(params.enable_thinking).toBe(true); + expect(params.preserve_thinking).toBeUndefined(); + }); }); diff --git a/packages/ai/test/issue-967-vision-guard.test.ts b/packages/ai/test/issue-967-vision-guard.test.ts index 6e7a55642..f060b916b 100644 --- a/packages/ai/test/issue-967-vision-guard.test.ts +++ b/packages/ai/test/issue-967-vision-guard.test.ts @@ -51,6 +51,7 @@ const compat: ResolvedOpenAICompat = { requiresReasoningContentForAllAssistantTurns: false, allowsSyntheticReasoningContentForToolCalls: true, replayReasoningContent: false, + qwenPreserveThinking: false, requiresAssistantContentForToolCalls: false, openRouterRouting: {}, vercelGatewayRouting: {}, diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 31f65e15c..b0206b630 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -185,6 +185,7 @@ describe("openai-completions compatibility", () => { requiresReasoningContentForAllAssistantTurns: false, allowsSyntheticReasoningContentForToolCalls: true, replayReasoningContent: false, + qwenPreserveThinking: false, requiresAssistantContentForToolCalls: false, openRouterRouting: {}, vercelGatewayRouting: {}, diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index 5af3f973a..5b6d76908 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -40,6 +40,7 @@ const compat: ResolvedOpenAICompat = { requiresReasoningContentForAllAssistantTurns: false, allowsSyntheticReasoningContentForToolCalls: true, replayReasoningContent: false, + qwenPreserveThinking: false, requiresAssistantContentForToolCalls: false, openRouterRouting: {}, vercelGatewayRouting: {}, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index f8e6356d8..072669fa5 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `OpenAICompat.qwenPreserveThinking` — auto-enabled when the resolved `thinkingFormat` is `"qwen"` or `"qwen-chat-template"` AND `replayReasoningContent` is on (i.e. the four built-in local OpenAI-compatible providers, or a custom provider pointed at a loopback / RFC1918 / `*.local` baseUrl). Pairs with the chat-completions encoder change so the request body carries `preserve_thinking: true` (twin top-level + `chat_template_kwargs` emission), keeping Qwen3.6+ from stripping `...` off older assistant turns and breaking the local slot's KV cache between user messages. Non-Qwen chat templates ignore the parameter, so the flag stays a no-op outside the Qwen path; users on a cloud Qwen host (Alibaba Dashscope / Qwen Portal) can opt in with `compat.qwenPreserveThinking: true`. ([#3541](https://github.com/can1357/oh-my-pi/issues/3541)) + ## [16.1.22] - 2026-06-26 ### Added diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 72feb366f..b6086c694 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -468,6 +468,20 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv replayReasoningContent: !PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) && (LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)), + // `preserve_thinking: true` makes the Qwen3.6+ chat template render + // `...` for older assistant turns too, instead of + // stripping it the moment a new user message moves them past + // `last_query_index`. Without it, the slot's KV cache (which holds the + // raw `X` tokens emitted during generation) diverges + // from the next-turn render and llama.cpp falls back to full prompt + // re-processing — the exact symptom reported in #3541. Auto-enabled + // for Qwen thinking dialects on local llama.cpp-style backends (paired + // with `replayReasoningContent` above). Non-Qwen templates ignore the + // parameter, so the flag stays a no-op outside the Qwen path. + qwenPreserveThinking: + (thinkingFormat === "qwen" || thinkingFormat === "qwen-chat-template") && + !PROXY_OPENAI_COMPAT_PROVIDERS.has(provider) && + (LOCAL_OPENAI_COMPAT_PROVIDERS.has(provider) || hasLocalLoopbackBaseUrl(baseUrl)), requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning, cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined, openRouterRouting: undefined, @@ -589,6 +603,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol // not via a top-level `reasoning_content` field — this flag is // chat-completions-only. replayReasoningContent: false, + // Responses-only; the Qwen `preserve_thinking` template knob lives on + // the chat-completions wire shape, never on Responses. + qwenPreserveThinking: false, requiresThinkingAsText: false, requiresMistralToolIds: false, requiresToolResultName: false, diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index b4d7b46de..45fd362bc 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -236,6 +236,28 @@ export interface OpenAICompat { * models). */ replayReasoningContent?: boolean; + /** + * Send `preserve_thinking: true` so the Qwen3.6+ chat template renders + * `...` markup for EVERY assistant turn (not just turns + * after the last user message). Without it, the template strips the think + * block from older assistant turns: + * + * ```jinja + * {%- if (preserve_thinking is defined and preserve_thinking is true) + * or (loop.index0 > ns.last_query_index) %} + * <|im_start|>assistant\n\n{rc}\n\n\n{content} + * {%- else %} + * <|im_start|>assistant\n{content} + * ``` + * + * The cache from the original generation has `...` tokens, + * so once a new user message arrives the prior assistant turns become + * "older" and the stripped re-render diverges — full prompt re-processing + * on SWA models (#3541). Default: auto-detected (Qwen thinking format on + * a local llama.cpp-style backend, paired with `replayReasoningContent`). + * Non-Qwen templates ignore the flag, so the auto-detection is safe. + */ + qwenPreserveThinking?: boolean; /** Whether assistant tool-call messages must include non-empty content. Default: false. */ requiresAssistantContentForToolCalls?: boolean; /** Whether the provider supports the `tool_choice` parameter. Default: true. */ @@ -448,6 +470,7 @@ export interface ResolvedOpenAISharedCompat { requiresReasoningContentForAllAssistantTurns: boolean; allowsSyntheticReasoningContentForToolCalls: boolean; replayReasoningContent: boolean; + qwenPreserveThinking: boolean; requiresThinkingAsText: boolean; requiresMistralToolIds: boolean; requiresToolResultName: boolean; @@ -498,6 +521,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & | "requiresReasoningContentForAllAssistantTurns" | "allowsSyntheticReasoningContentForToolCalls" | "replayReasoningContent" + | "qwenPreserveThinking" | "requiresThinkingAsText" | "requiresMistralToolIds" | "requiresToolResultName"