From ec2caea840b8a23e84f6fd7289d5e97e5c04f1e6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 30 Jul 2026 13:19:41 +0000 Subject: [PATCH] fix(coding-agent): honor remote llama.cpp/ollama discovery base urls llama.cpp and Ollama model discovery probed /models and /props with a 250ms timeout tuned for a loopback server. That cap also applied to a host reached over the network, so a remote or LAN LLAMA_CPP_BASE_URL (or OLLAMA_BASE_URL/OLLAMA_HOST) with normal round-trip latency timed out, discovery returned no models, and the picker fell back to stale 127.0.0.1:8080 entries. Select the probe timeout by host: strictly-loopback base URLs keep the fast fail so a busy or foreign service on the default port never stalls startup; every non-loopback host gets a generous discovery budget. Fixes #7087 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/config/model-discovery.ts | 37 ++++++++++++++++--- .../coding-agent/test/model-discovery.test.ts | 32 +++++++++++++++- 3 files changed, 67 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 74bcb76e0..982b7006b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed a remote or LAN `LLAMA_CPP_BASE_URL` (and `OLLAMA_BASE_URL`/`OLLAMA_HOST`) being ignored during model discovery: the `/models` and `/props` probes used a 250ms timeout tuned for a loopback server, so a host reached over the network exceeded it, discovery returned no models, and the picker fell back to stale `127.0.0.1:8080` entries. Non-loopback hosts now get a generous discovery timeout while the loopback default keeps its fast fail ([#7087](https://github.com/can1357/oh-my-pi/issues/7087)). + ## [17.2.0] - 2026-07-30 ### Breaking Changes diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 2f7335c16..bc5004240 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -62,6 +62,33 @@ async function withTimeoutSignal(timeoutMs: number, fn: (signal: AbortSignal) } } +/** Generous discovery budget for a non-loopback (remote / LAN) inference host. */ +const REMOTE_DISCOVERY_TIMEOUT_MS = 10_000; + +/** + * Pick a discovery-probe timeout for a local-engine base URL. + * + * The implicit `127.0.0.1` default probe keeps a tight `loopbackMs` cap so a + * busy or foreign service on the default port never stalls startup. But that + * cap is far too short for a host reached over the network: a user who points + * `LLAMA_CPP_BASE_URL` / `OLLAMA_BASE_URL` / `OLLAMA_HOST` at a remote or LAN + * machine has real round-trip latency, and a 250ms cap made that server look + * empty (issue #7087). Anything that is not strictly loopback therefore gets + * {@link REMOTE_DISCOVERY_TIMEOUT_MS}. + */ +export function discoveryProbeTimeoutMs(baseUrl: string, loopbackMs: number): number { + let hostname: string; + try { + hostname = new URL(baseUrl).hostname; + } catch { + return loopbackMs; + } + hostname = hostname.replace(/^\[/, "").replace(/\]$/, ""); + const isLoopback = + hostname === "localhost" || hostname === "0.0.0.0" || hostname === "::1" || /^127\./.test(hostname); + return isLoopback ? loopbackMs : REMOTE_DISCOVERY_TIMEOUT_MS; +} + const DEFAULT_OLLAMA_BASE_URL = "http://127.0.0.1:11434"; const OLLAMA_HOST_DEFAULT_PORT = "11434"; @@ -391,7 +418,7 @@ async function discoverOllamaModelMetadata( ): Promise { const showUrl = `${endpoint}/api/show`; try { - const payload = await withTimeoutSignal(150, async signal => { + const payload = await withTimeoutSignal(discoveryProbeTimeoutMs(endpoint, 150), async signal => { const response = await ctx.fetch(showUrl, { method: "POST", headers: { ...(headers ?? {}), "Content-Type": "application/json" }, @@ -444,7 +471,7 @@ export async function discoverOllamaModels( const endpoint = normalizeOllamaBaseUrl(providerConfig.baseUrl); const tagsUrl = `${endpoint}/api/tags`; const headers = { ...(providerConfig.headers ?? {}) }; - const payload = await withTimeoutSignal(250, async signal => { + const payload = await withTimeoutSignal(discoveryProbeTimeoutMs(endpoint, 250), async signal => { const response = await ctx.fetch(tagsUrl, { headers, signal, @@ -491,7 +518,7 @@ async function discoverLlamaCppServerMetadata( ): Promise { const propsUrl = `${toLlamaCppNativeBaseUrl(baseUrl)}/props`; try { - const payload = await withTimeoutSignal(150, async signal => { + const payload = await withTimeoutSignal(discoveryProbeTimeoutMs(baseUrl, 150), async signal => { const response = await ctx.fetch(propsUrl, { headers, signal, @@ -567,7 +594,7 @@ export async function discoverLlamaCppModels( let headers = baseHeaders; const attempt = async (h: Record) => { const [payload, metadata] = await Promise.all([ - withTimeoutSignal(250, async signal => { + withTimeoutSignal(discoveryProbeTimeoutMs(baseUrl, 250), async signal => { const response = await ctx.fetch(modelsUrl, { headers: h, signal, @@ -641,7 +668,7 @@ export async function discoverLlamaCppModelRuntimeMetadata( const baseHeaders: Record = { ...(model.headers ?? {}) }; const attempt = async (headers: Record) => { const [entries, serverMetadata] = await Promise.all([ - withTimeoutSignal(250, async signal => { + withTimeoutSignal(discoveryProbeTimeoutMs(nativeBaseUrl, 250), async signal => { const response = await ctx.fetch(modelsUrl, { headers, signal, diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index 169d3582c..73a0dc0d5 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -9,7 +9,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types"; -import { applyLlamaCppQwenThinking } from "@oh-my-pi/pi-coding-agent/config/model-discovery"; +import { applyLlamaCppQwenThinking, discoveryProbeTimeoutMs } from "@oh-my-pi/pi-coding-agent/config/model-discovery"; import { kNoAuth, ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { resetSettingsForTest } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; @@ -1203,6 +1203,36 @@ describe("ModelRegistry runtime discovery", () => { expect(requested).not.toContain("http://127.0.0.1:8080/v1/models"); }); + test("discoveryProbeTimeoutMs keeps loopback fast but gives non-loopback hosts a larger budget", () => { + // Regression: the loopback-tuned probe timeout was applied to every host, + // so a remote/LAN LLAMA_CPP_BASE_URL with normal round-trip latency timed + // out and the model list came back empty (#7087). Loopback keeps the tight + // budget; anything reached over the network gets a strictly larger one. + const loopbackMs = 250; + for (const host of [ + "http://127.0.0.1:8080", + "http://127.5.6.7:8080", + "http://localhost:8080", + "http://[::1]:8080", + "http://0.0.0.0:8080", + ]) { + expect(discoveryProbeTimeoutMs(host, loopbackMs)).toBe(loopbackMs); + } + const remoteBudgets = [ + "http://remote-llama.test:8080", + "http://192.168.1.50:8080", + "http://172.18.0.3:8080", + "http://10.0.0.4:8080", + "http://box.local:8080", + ].map(host => discoveryProbeTimeoutMs(host, loopbackMs)); + for (const budget of remoteBudgets) { + expect(budget).toBeGreaterThan(loopbackMs); + } + // A consistent budget for every non-loopback host, independent of the tight cap. + expect(new Set(remoteBudgets).size).toBe(1); + expect(discoveryProbeTimeoutMs("http://remote-llama.test:8080", 150)).toBe(remoteBudgets[0]); + }); + test("llama.cpp discovery marks per-model architecture image modalities as vision-capable", async () => { const fetchMock: FetchImpl = async input => { const url = String(input);