fix(coding-agent): auto-enable append-only context for Ollama and local servers
Ollama, LM Studio, and llama.cpp / vLLM all do byte-prefix KV cache reuse on the model server side. The agent loop's non-append-only path rebuilds the system prompt and tool catalogue on every turn through fresh allocations (`normalizeTools`, `convertToLlm`, optional memory-backend `beforeAgentStartPrompt` injection), which dirties enough leading bytes to invalidate the cache and force a full prompt re-evaluation. `shouldAutoEnableAppendOnlyContext` previously only recognized DeepSeek and Xiaomi Token Plan. Extend the auto-detect: - Allowlist the known local-server provider ids (`ollama`, `ollama-cloud`, `lm-studio`). - Detect user-defined local servers by parsing `baseUrl`: loopback, RFC1918 private IPv4, and `.local` mDNS hostnames. - Keep the existing `compat.supportsStore` opt-in escape hatch and the explicit `provider.appendOnlyContext: on`/`off` override paths. Regression coverage in `append-only-context-mode.test.ts` exercises each new positive case plus negative samples (172.15/172.32 just outside RFC1918, public hosts, malformed URLs). Fixes #3033
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama (and LM Studio / loopback llama.cpp / vLLM servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)).
|
||||
|
||||
## [16.1.1] - 2026-06-19
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -8,10 +8,53 @@ export interface AppendOnlyContextModel {
|
||||
compatConfig?: object;
|
||||
}
|
||||
|
||||
/**
|
||||
* Local model servers (Ollama, LM Studio, llama.cpp, vLLM, sglang, …) all
|
||||
* rely on llama.cpp-style prefix KV-cache reuse: identical leading tokens
|
||||
* skip re-prefill on the next request. Append-only mode is the only way to
|
||||
* guarantee byte-stable bytes across turns, since the live system prompt,
|
||||
* tool catalogue, and message log all flow through fresh allocations every
|
||||
* step (see `agent-loop.ts` `streamAssistantResponse` fallback path).
|
||||
*/
|
||||
const LOCAL_INFERENCE_PROVIDERS = new Set(["ollama", "ollama-cloud", "lm-studio"]);
|
||||
|
||||
/** True when `baseUrl` resolves to a loopback or RFC1918 host — the heuristic
|
||||
* for user-defined providers (llama.cpp / vLLM via `models.yaml`) that omp
|
||||
* has no provider id for. Substring match on parsed hostname only; ports,
|
||||
* paths, and unparseable URLs return false.
|
||||
*/
|
||||
function hasLocalLoopbackBaseUrl(baseUrl: string | undefined): boolean {
|
||||
if (!baseUrl) return false;
|
||||
let hostname: string;
|
||||
try {
|
||||
hostname = new URL(baseUrl).hostname.toLowerCase();
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
if (
|
||||
hostname === "localhost" ||
|
||||
hostname === "127.0.0.1" ||
|
||||
hostname === "0.0.0.0" ||
|
||||
hostname === "::1" ||
|
||||
hostname === "[::1]"
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
// RFC1918 private IPv4 ranges.
|
||||
if (/^10\./.test(hostname)) return true;
|
||||
if (/^192\.168\./.test(hostname)) return true;
|
||||
if (/^172\.(1[6-9]|2[0-9]|3[01])\./.test(hostname)) return true;
|
||||
// Common ".local" mDNS hostnames used for home-LAN llama.cpp boxes.
|
||||
if (hostname.endsWith(".local")) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean {
|
||||
if (!model) return false;
|
||||
if (model.provider === "deepseek") return true;
|
||||
if (LOCAL_INFERENCE_PROVIDERS.has(model.provider)) return true;
|
||||
if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true;
|
||||
if (hasLocalLoopbackBaseUrl(model.baseUrl)) return true;
|
||||
return !!model.compatConfig && "supportsStore" in model.compatConfig && model.compatConfig.supportsStore === true;
|
||||
}
|
||||
|
||||
|
||||
@@ -41,4 +41,46 @@ describe("shouldEnableAppendOnlyContext", () => {
|
||||
test("auto remains off for unknown providers without prefix-cache signals", () => {
|
||||
expect(shouldEnableAppendOnlyContext("auto", GENERIC_PROXY)).toBe(false);
|
||||
});
|
||||
|
||||
test("auto enables for local inference providers", () => {
|
||||
// Ollama serves both `ollama-chat` (cloud-managed) and the openai-responses
|
||||
// path used by locally pulled models — issue #3033 (llama.cpp KV-cache prefix
|
||||
// resets every turn without append-only mode).
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama", baseUrl: "http://127.0.0.1:11434" })).toBe(
|
||||
true,
|
||||
);
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama-cloud", baseUrl: "https://ollama.com" })).toBe(
|
||||
true,
|
||||
);
|
||||
expect(
|
||||
shouldEnableAppendOnlyContext("auto", { provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
test("auto enables for loopback and private baseUrls (user-defined llama.cpp/vLLM)", () => {
|
||||
const cases: Array<{ provider: string; baseUrl: string }> = [
|
||||
{ provider: "my-llamacpp", baseUrl: "http://localhost:8080/v1" },
|
||||
{ provider: "my-vllm", baseUrl: "http://127.0.0.1:8000/v1" },
|
||||
{ provider: "my-sglang", baseUrl: "http://[::1]:30000/v1" },
|
||||
{ provider: "lan-host", baseUrl: "http://192.168.1.42:11434" },
|
||||
{ provider: "lan-host", baseUrl: "http://10.0.0.5:11434" },
|
||||
{ provider: "lan-host", baseUrl: "http://172.17.0.3:11434" },
|
||||
{ provider: "mdns-host", baseUrl: "http://gpu-box.local:11434" },
|
||||
];
|
||||
for (const model of cases) {
|
||||
expect(shouldEnableAppendOnlyContext("auto", model)).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
test("auto stays off for public hosts that merely share an IP prefix", () => {
|
||||
// 172.15.x.x sits just outside the RFC1918 16-31 band.
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.15.0.1/v1" })).toBe(false);
|
||||
// 172.32.x.x sits just outside the RFC1918 16-31 band on the other side.
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.32.0.1/v1" })).toBe(false);
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "https://example.com/v1" })).toBe(false);
|
||||
});
|
||||
|
||||
test("malformed baseUrl never crashes the resolver", () => {
|
||||
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "not a url" })).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user