From 4353920b95909f36e85d7121e3789e05fe4e85b5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 19 Jun 2026 08:26:14 +0000 Subject: [PATCH] fix(coding-agent): include llama.cpp in append-only allowlist ModelRegistry registers a built-in llama.cpp discovery as `provider: "llama.cpp"` (model-registry.ts:1063), so a reverse-proxied or public-DNS llama.cpp endpoint never trips the loopback heuristic and was still falling through to no append-only context. Add "llama.cpp" to LOCAL_INFERENCE_PROVIDERS, refresh the docstring on `hasLocalLoopbackBaseUrl` (it covers user-defined providers; built- in local ids are caught by the allowlist), and extend the test to exercise a public-host llama.cpp baseUrl so the allowlist path is covered independently of the loopback heuristic. Refs #3033 --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/config/append-only-context-mode.ts | 12 +++++++----- .../test/append-only-context-mode.test.ts | 8 ++++++++ 3 files changed, 16 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index dce09b2eb..f9b92b438 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Ollama (and LM Studio / loopback llama.cpp / vLLM servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)). +- Fixed Ollama, LM Studio, and llama.cpp (plus loopback vLLM / sglang servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, `llama.cpp`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)). ## [16.1.1] - 2026-06-19 diff --git a/packages/coding-agent/src/config/append-only-context-mode.ts b/packages/coding-agent/src/config/append-only-context-mode.ts index d0e00707d..0856a0f93 100644 --- a/packages/coding-agent/src/config/append-only-context-mode.ts +++ b/packages/coding-agent/src/config/append-only-context-mode.ts @@ -16,12 +16,14 @@ export interface AppendOnlyContextModel { * tool catalogue, and message log all flow through fresh allocations every * step (see `agent-loop.ts` `streamAssistantResponse` fallback path). */ -const LOCAL_INFERENCE_PROVIDERS = new Set(["ollama", "ollama-cloud", "lm-studio"]); +const LOCAL_INFERENCE_PROVIDERS = new Set(["ollama", "ollama-cloud", "lm-studio", "llama.cpp"]); -/** True when `baseUrl` resolves to a loopback or RFC1918 host — the heuristic - * for user-defined providers (llama.cpp / vLLM via `models.yaml`) that omp - * has no provider id for. Substring match on parsed hostname only; ports, - * paths, and unparseable URLs return false. +/** True when `baseUrl` resolves to a loopback or RFC1918 host — covers + * llama.cpp/vLLM/sglang servers registered under a user-defined provider id + * via `models.yaml`. Built-in local provider ids (`ollama`, `lm-studio`, + * `llama.cpp`) are already handled by `LOCAL_INFERENCE_PROVIDERS`. + * Substring match on the parsed hostname only; ports, paths, and unparseable + * URLs return false. */ function hasLocalLoopbackBaseUrl(baseUrl: string | undefined): boolean { if (!baseUrl) return false; diff --git a/packages/coding-agent/test/append-only-context-mode.test.ts b/packages/coding-agent/test/append-only-context-mode.test.ts index 32550ef22..6c1727893 100644 --- a/packages/coding-agent/test/append-only-context-mode.test.ts +++ b/packages/coding-agent/test/append-only-context-mode.test.ts @@ -55,6 +55,14 @@ describe("shouldEnableAppendOnlyContext", () => { expect( shouldEnableAppendOnlyContext("auto", { provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }), ).toBe(true); + // `llama.cpp` is a built-in provider id (ModelRegistry registers it for keyless local discovery); + // the allowlist must catch it even when the user reverse-proxies the server through a public host. + expect( + shouldEnableAppendOnlyContext("auto", { provider: "llama.cpp", baseUrl: "https://llamacpp.example.com/v1" }), + ).toBe(true); + expect(shouldEnableAppendOnlyContext("auto", { provider: "llama.cpp", baseUrl: "http://127.0.0.1:8080" })).toBe( + true, + ); }); test("auto enables for loopback and private baseUrls (user-defined llama.cpp/vLLM)", () => {