fix(coding-agent): auto-enable append-only context for Ollama and local servers

Ollama, LM Studio, and llama.cpp / vLLM all do byte-prefix KV cache reuse
on the model server side. The agent loop's non-append-only path rebuilds
the system prompt and tool catalogue on every turn through fresh
allocations (`normalizeTools`, `convertToLlm`, optional memory-backend
`beforeAgentStartPrompt` injection), which dirties enough leading bytes
to invalidate the cache and force a full prompt re-evaluation.

`shouldAutoEnableAppendOnlyContext` previously only recognized DeepSeek
and Xiaomi Token Plan. Extend the auto-detect:

- Allowlist the known local-server provider ids (`ollama`,
  `ollama-cloud`, `lm-studio`).
- Detect user-defined local servers by parsing `baseUrl`: loopback,
  RFC1918 private IPv4, and `.local` mDNS hostnames.
- Keep the existing `compat.supportsStore` opt-in escape hatch and the
  explicit `provider.appendOnlyContext: on`/`off` override paths.

Regression coverage in `append-only-context-mode.test.ts` exercises
each new positive case plus negative samples (172.15/172.32 just outside
RFC1918, public hosts, malformed URLs).

Fixes #3033
This commit is contained in:
roboomp
2026-06-19 08:20:15 +00:00
parent 492fe5e896
commit 9a07533924
3 changed files with 89 additions and 0 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Ollama (and LM Studio / loopback llama.cpp / vLLM servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)).
## [16.1.1] - 2026-06-19
### Changed
@@ -8,10 +8,53 @@ export interface AppendOnlyContextModel {
compatConfig?: object;
}
/**
* Local model servers (Ollama, LM Studio, llama.cpp, vLLM, sglang, …) all
* rely on llama.cpp-style prefix KV-cache reuse: identical leading tokens
* skip re-prefill on the next request. Append-only mode is the only way to
* guarantee byte-stable bytes across turns, since the live system prompt,
* tool catalogue, and message log all flow through fresh allocations every
* step (see `agent-loop.ts` `streamAssistantResponse` fallback path).
*/
const LOCAL_INFERENCE_PROVIDERS = new Set(["ollama", "ollama-cloud", "lm-studio"]);
/** True when `baseUrl` resolves to a loopback or RFC1918 host — the heuristic
* for user-defined providers (llama.cpp / vLLM via `models.yaml`) that omp
* has no provider id for. Substring match on parsed hostname only; ports,
* paths, and unparseable URLs return false.
*/
function hasLocalLoopbackBaseUrl(baseUrl: string | undefined): boolean {
if (!baseUrl) return false;
let hostname: string;
try {
hostname = new URL(baseUrl).hostname.toLowerCase();
} catch {
return false;
}
if (
hostname === "localhost" ||
hostname === "127.0.0.1" ||
hostname === "0.0.0.0" ||
hostname === "::1" ||
hostname === "[::1]"
) {
return true;
}
// RFC1918 private IPv4 ranges.
if (/^10\./.test(hostname)) return true;
if (/^192\.168\./.test(hostname)) return true;
if (/^172\.(1[6-9]|2[0-9]|3[01])\./.test(hostname)) return true;
// Common ".local" mDNS hostnames used for home-LAN llama.cpp boxes.
if (hostname.endsWith(".local")) return true;
return false;
}
function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean {
if (!model) return false;
if (model.provider === "deepseek") return true;
if (LOCAL_INFERENCE_PROVIDERS.has(model.provider)) return true;
if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true;
if (hasLocalLoopbackBaseUrl(model.baseUrl)) return true;
return !!model.compatConfig && "supportsStore" in model.compatConfig && model.compatConfig.supportsStore === true;
}
@@ -41,4 +41,46 @@ describe("shouldEnableAppendOnlyContext", () => {
test("auto remains off for unknown providers without prefix-cache signals", () => {
expect(shouldEnableAppendOnlyContext("auto", GENERIC_PROXY)).toBe(false);
});
test("auto enables for local inference providers", () => {
// Ollama serves both `ollama-chat` (cloud-managed) and the openai-responses
// path used by locally pulled models — issue #3033 (llama.cpp KV-cache prefix
// resets every turn without append-only mode).
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama", baseUrl: "http://127.0.0.1:11434" })).toBe(
true,
);
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama-cloud", baseUrl: "https://ollama.com" })).toBe(
true,
);
expect(
shouldEnableAppendOnlyContext("auto", { provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }),
).toBe(true);
});
test("auto enables for loopback and private baseUrls (user-defined llama.cpp/vLLM)", () => {
const cases: Array<{ provider: string; baseUrl: string }> = [
{ provider: "my-llamacpp", baseUrl: "http://localhost:8080/v1" },
{ provider: "my-vllm", baseUrl: "http://127.0.0.1:8000/v1" },
{ provider: "my-sglang", baseUrl: "http://[::1]:30000/v1" },
{ provider: "lan-host", baseUrl: "http://192.168.1.42:11434" },
{ provider: "lan-host", baseUrl: "http://10.0.0.5:11434" },
{ provider: "lan-host", baseUrl: "http://172.17.0.3:11434" },
{ provider: "mdns-host", baseUrl: "http://gpu-box.local:11434" },
];
for (const model of cases) {
expect(shouldEnableAppendOnlyContext("auto", model)).toBe(true);
}
});
test("auto stays off for public hosts that merely share an IP prefix", () => {
// 172.15.x.x sits just outside the RFC1918 16-31 band.
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.15.0.1/v1" })).toBe(false);
// 172.32.x.x sits just outside the RFC1918 16-31 band on the other side.
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.32.0.1/v1" })).toBe(false);
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "https://example.com/v1" })).toBe(false);
});
test("malformed baseUrl never crashes the resolver", () => {
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "not a url" })).toBe(false);
});
});