From 9b496847235f16c1804ce65583cd8c4f29f6faaa Mon Sep 17 00:00:00 2001 From: Chris Phlipot Date: Thu, 13 Aug 2026 11:31:27 -0700 Subject: [PATCH 1/4] use /v1/ APIs for llama.cpp for better compatibility Llama.cpp mirrors its /v1/ apis to / which omp currently uses, however this change is somewhat recent of only a few months ago, so omp's llama.cpp provider does not work with older versions of llama.cpp and some forks. to improve compatbility use /v1/ apis for requests. model discovery still uses /models and /props directly without v1. because modern versions of llama.cpp mirror these, people using recent versions should see no impact from this, while people using older version should see improved compatibility. --- packages/coding-agent/src/config/model-discovery.ts | 2 +- packages/coding-agent/test/model-discovery.test.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 5caafa822..9a5c8a285 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -645,7 +645,7 @@ export async function discoverLlamaCppModels( name: id, api: providerConfig.api, provider: providerConfig.provider, - baseUrl, + baseUrl: ensureLlamaCppV1BaseUrl(baseUrl), reasoning: false, input: item.input ?? serverMetadata?.input ?? ["text"], imageInputDecoder: "stb", diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index c25d30010..f86b79d77 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -1158,7 +1158,7 @@ describe("ModelRegistry runtime discovery", () => { const plain = registry.find("llama.cpp", "llama-3.1-8b"); expect(plain?.reasoning).toBe(false); expect(plain?.api).toBe("openai-responses"); - expect(plain?.baseUrl).toBe("http://127.0.0.1:8080"); + expect(plain?.baseUrl).toBe("http://127.0.0.1:8080/v1"); expect((plain?.compat as DialectFields | undefined)?.reasoningDisableMode).not.toBe("qwen-template-false"); }); From 617fc4d51b858dd9342b5146c07dcc5cedfd111c Mon Sep 17 00:00:00 2001 From: Chris Phlipot Date: Thu, 13 Aug 2026 11:52:17 -0700 Subject: [PATCH 2/4] update changelog --- packages/coding-agent/CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4542af48d..a5c31b53d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- Fixed llama.cpp model discovery producing `baseUrl` without the `/v1` prefix for non-Qwen models, causing 404 errors on OpenAI-compatible endpoints (`/v1/responses`, `/v1/chat/completions`). The fix ensures all discovered llama.cpp models include `/v1` in their `baseUrl`, matching the behavior already applied to Qwen models. ## [17.3.1] - 2026-08-13 From 23319fa413679a6ee6c625f1f02227e67442c496 Mon Sep 17 00:00:00 2001 From: Chris Phlipot Date: Thu, 13 Aug 2026 20:35:40 -0700 Subject: [PATCH 3/4] fixup /v1/ for all llama.cpp models, not just qwen. --- .../coding-agent/src/config/model-registry.ts | 31 ++++++++++--------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index f2dd6f1b5..5f7054f3c 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -577,7 +577,7 @@ export class ModelRegistry { const withConfigModels = this.#mergeCustomModels(resolvedDefaults, select(this.#customModelOverlays)); const combined = this.#mergeCustomModels(withConfigModels, select(this.#runtimeModelOverlays)); const withModelOverrides = this.#applyModelOverrides(collapseBuiltModelVariants(combined), this.#modelOverrides); - return this.#applyLlamaCppQwenThinkingToModels(this.#applyRuntimeProviderOverrides(withModelOverrides)); + return this.#applyLlamaCppModelFixups(this.#applyRuntimeProviderOverrides(withModelOverrides)); } #composeStaticModels(providerFilter?: ReadonlySet): Model[] { @@ -1064,9 +1064,7 @@ export class ModelRegistry { const withConfigModels = this.#mergeCustomModels(resolved, this.#customModelOverlays); const combined = this.#mergeCustomModels(withConfigModels, this.#runtimeModelOverlays); const withModelOverrides = this.#applyModelOverrides(collapseBuiltModelVariants(combined), this.#modelOverrides); - this.#unprojectedModels = this.#applyLlamaCppQwenThinkingToModels( - this.#applyRuntimeProviderOverrides(withModelOverrides), - ); + this.#unprojectedModels = this.#applyLlamaCppModelFixups(this.#applyRuntimeProviderOverrides(withModelOverrides)); this.#models = this.#applyRuntimeModelModifiers(this.#unprojectedModels); } @@ -1446,20 +1444,25 @@ export class ModelRegistry { }); } - // #applyLlamaCppQwenThinkingToModels re-runs applyLlamaCppQwenThinking as the - // outermost transform for llama.cpp-provider models, after discovery merges, - // cache fallbacks, and provider/transport overrides have run. It is - // idempotent, so it restores the routed Qwen model's chat-completions api, - // `/v1` runtime base URL, and disable dialect even when a configured `baseUrl` - // override (which wins in mergeDiscoveredModel) or a fallback to a pre-fix - // cached row would otherwise leave the old spec in place. - #applyLlamaCppQwenThinkingToModels(models: Model[]): Model[] { + // #applyLlamaCppModelFixups is the outermost transform for llama.cpp-provider + // models, after discovery merges, cache fallbacks, and provider/transport + // overrides have run. It applies Qwen-specific fixes (api, reasoning, compat) + // and ensures all non-transport models have the `/v1` prefix in their baseUrl, + // even when a configured override or stale cache row would strip it. + #applyLlamaCppModelFixups(models: Model[]): Model[] { const llamaCppProviders = new Set(); for (const provider of this.#discoverableProviders) { if (provider.discovery.type === "llama.cpp") llamaCppProviders.add(provider.provider); } if (llamaCppProviders.size === 0) return models; - return models.map(model => (llamaCppProviders.has(model.provider) ? applyLlamaCppQwenThinking(model) : model)); + return models.map(model => { + if (!llamaCppProviders.has(model.provider)) return model; + const withFixups = applyLlamaCppQwenThinking(model); + if (!withFixups.transport && !withFixups.baseUrl.endsWith("/v1")) { + return buildModel({ ...withFixups, baseUrl: `${withFixups.baseUrl}/v1` }); + } + return withFixups; + }); } #mergeProviderOverride(baseOverride: ProviderOverride | undefined, override: ProviderOverride): ProviderOverride { @@ -2097,7 +2100,7 @@ export class ModelRegistry { transportOverride, ); this.#runtimeProviderOverrides.set(providerName, nextRuntimeOverride); - this.#unprojectedModels = this.#applyLlamaCppQwenThinkingToModels( + this.#unprojectedModels = this.#applyLlamaCppModelFixups( this.#unprojectedModels.map(model => { if (model.provider !== providerName) return model; return this.#applyProviderTransportOverrideToModel(model, transportOverride); From 6ebc4042e6394738077c9242b25a6b1f6c9a4006 Mon Sep 17 00:00:00 2001 From: Chris Phlipot Date: Thu, 13 Aug 2026 21:13:37 -0700 Subject: [PATCH 4/4] reuse existing functions to ensure v1 prefix --- packages/coding-agent/src/config/model-discovery.ts | 4 ++-- packages/coding-agent/src/config/model-registry.ts | 7 ++++++- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 9a5c8a285..ca0f2427e 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -1018,7 +1018,7 @@ export async function discoverProxyModels( return discovered; } -function normalizeLlamaCppBaseUrl(baseUrl?: string): string { +export function normalizeLlamaCppBaseUrl(baseUrl?: string): string { const defaultBaseUrl = "http://127.0.0.1:8080"; const raw = baseUrl || defaultBaseUrl; try { @@ -1033,7 +1033,7 @@ function normalizeLlamaCppBaseUrl(baseUrl?: string): string { // ensureLlamaCppV1BaseUrl appends the OpenAI-compatible `/v1` prefix a // chat-completions request needs; native discovery keeps the bare root, which // serves `/models` and `/props` but not `/chat/completions`. -function ensureLlamaCppV1BaseUrl(baseUrl: string): string { +export function ensureLlamaCppV1BaseUrl(baseUrl: string): string { return baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`; } diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 5f7054f3c..456ce6cd9 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -61,9 +61,11 @@ import { type DiscoveryProviderConfig, discoverLlamaCppModelRuntimeMetadata, discoverModelsByProviderType, + ensureLlamaCppV1BaseUrl, getImplicitOllamaBaseUrl, getOllamaContextLengthOverride, normalizeLiteLLMDiscoveryBaseUrl, + normalizeLlamaCppBaseUrl, } from "./model-discovery"; import { AUTHORITATIVE_RUNTIME_CATALOG_PROVIDERS, @@ -1459,7 +1461,10 @@ export class ModelRegistry { if (!llamaCppProviders.has(model.provider)) return model; const withFixups = applyLlamaCppQwenThinking(model); if (!withFixups.transport && !withFixups.baseUrl.endsWith("/v1")) { - return buildModel({ ...withFixups, baseUrl: `${withFixups.baseUrl}/v1` }); + return buildModel({ + ...withFixups, + baseUrl: ensureLlamaCppV1BaseUrl(normalizeLlamaCppBaseUrl(withFixups.baseUrl)), + }); } return withFixups; });