Merge PR #8459: fix(coding-agent): use /v1/ APIs for llama.cpp for better compatibility (@cphlipot)

This commit is contained in:
can1357
2026-08-16 02:13:37 +02:00
4 changed files with 27 additions and 18 deletions
+1
View File
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
- Fixed llama.cpp model discovery producing `baseUrl` without the `/v1` prefix for non-Qwen models, causing 404 errors on OpenAI-compatible endpoints (`/v1/responses`, `/v1/chat/completions`). The fix ensures all discovered llama.cpp models include `/v1` in their `baseUrl`, matching the behavior already applied to Qwen models.
### Changed
@@ -645,7 +645,7 @@ export async function discoverLlamaCppModels(
name: id,
api: providerConfig.api,
provider: providerConfig.provider,
baseUrl,
baseUrl: ensureLlamaCppV1BaseUrl(baseUrl),
reasoning: false,
input: item.input ?? serverMetadata?.input ?? ["text"],
imageInputDecoder: "stb",
@@ -1018,7 +1018,7 @@ export async function discoverProxyModels(
return discovered;
}
function normalizeLlamaCppBaseUrl(baseUrl?: string): string {
export function normalizeLlamaCppBaseUrl(baseUrl?: string): string {
const defaultBaseUrl = "http://127.0.0.1:8080";
const raw = baseUrl || defaultBaseUrl;
try {
@@ -1033,7 +1033,7 @@ function normalizeLlamaCppBaseUrl(baseUrl?: string): string {
// ensureLlamaCppV1BaseUrl appends the OpenAI-compatible `/v1` prefix a
// chat-completions request needs; native discovery keeps the bare root, which
// serves `/models` and `/props` but not `/chat/completions`.
function ensureLlamaCppV1BaseUrl(baseUrl: string): string {
export function ensureLlamaCppV1BaseUrl(baseUrl: string): string {
return baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`;
}
@@ -61,9 +61,11 @@ import {
type DiscoveryProviderConfig,
discoverLlamaCppModelRuntimeMetadata,
discoverModelsByProviderType,
ensureLlamaCppV1BaseUrl,
getImplicitOllamaBaseUrl,
getOllamaContextLengthOverride,
normalizeLiteLLMDiscoveryBaseUrl,
normalizeLlamaCppBaseUrl,
} from "./model-discovery";
import {
AUTHORITATIVE_RUNTIME_CATALOG_PROVIDERS,
@@ -577,7 +579,7 @@ export class ModelRegistry {
const withConfigModels = this.#mergeCustomModels(resolvedDefaults, select(this.#customModelOverlays));
const combined = this.#mergeCustomModels(withConfigModels, select(this.#runtimeModelOverlays));
const withModelOverrides = this.#applyModelOverrides(collapseBuiltModelVariants(combined), this.#modelOverrides);
return this.#applyLlamaCppQwenThinkingToModels(this.#applyRuntimeProviderOverrides(withModelOverrides));
return this.#applyLlamaCppModelFixups(this.#applyRuntimeProviderOverrides(withModelOverrides));
}
#composeStaticModels(providerFilter?: ReadonlySet<string>): Model<Api>[] {
@@ -1064,9 +1066,7 @@ export class ModelRegistry {
const withConfigModels = this.#mergeCustomModels(resolved, this.#customModelOverlays);
const combined = this.#mergeCustomModels(withConfigModels, this.#runtimeModelOverlays);
const withModelOverrides = this.#applyModelOverrides(collapseBuiltModelVariants(combined), this.#modelOverrides);
this.#unprojectedModels = this.#applyLlamaCppQwenThinkingToModels(
this.#applyRuntimeProviderOverrides(withModelOverrides),
);
this.#unprojectedModels = this.#applyLlamaCppModelFixups(this.#applyRuntimeProviderOverrides(withModelOverrides));
this.#models = this.#applyRuntimeModelModifiers(this.#unprojectedModels);
}
@@ -1446,20 +1446,28 @@ export class ModelRegistry {
});
}
// #applyLlamaCppQwenThinkingToModels re-runs applyLlamaCppQwenThinking as the
// outermost transform for llama.cpp-provider models, after discovery merges,
// cache fallbacks, and provider/transport overrides have run. It is
// idempotent, so it restores the routed Qwen model's chat-completions api,
// `/v1` runtime base URL, and disable dialect even when a configured `baseUrl`
// override (which wins in mergeDiscoveredModel) or a fallback to a pre-fix
// cached row would otherwise leave the old spec in place.
#applyLlamaCppQwenThinkingToModels(models: Model<Api>[]): Model<Api>[] {
// #applyLlamaCppModelFixups is the outermost transform for llama.cpp-provider
// models, after discovery merges, cache fallbacks, and provider/transport
// overrides have run. It applies Qwen-specific fixes (api, reasoning, compat)
// and ensures all non-transport models have the `/v1` prefix in their baseUrl,
// even when a configured override or stale cache row would strip it.
#applyLlamaCppModelFixups(models: Model<Api>[]): Model<Api>[] {
const llamaCppProviders = new Set<string>();
for (const provider of this.#discoverableProviders) {
if (provider.discovery.type === "llama.cpp") llamaCppProviders.add(provider.provider);
}
if (llamaCppProviders.size === 0) return models;
return models.map(model => (llamaCppProviders.has(model.provider) ? applyLlamaCppQwenThinking(model) : model));
return models.map(model => {
if (!llamaCppProviders.has(model.provider)) return model;
const withFixups = applyLlamaCppQwenThinking(model);
if (!withFixups.transport && !withFixups.baseUrl.endsWith("/v1")) {
return buildModel({
...withFixups,
baseUrl: ensureLlamaCppV1BaseUrl(normalizeLlamaCppBaseUrl(withFixups.baseUrl)),
});
}
return withFixups;
});
}
#mergeProviderOverride(baseOverride: ProviderOverride | undefined, override: ProviderOverride): ProviderOverride {
@@ -2102,7 +2110,7 @@ export class ModelRegistry {
transportOverride,
);
this.#runtimeProviderOverrides.set(providerName, nextRuntimeOverride);
this.#unprojectedModels = this.#applyLlamaCppQwenThinkingToModels(
this.#unprojectedModels = this.#applyLlamaCppModelFixups(
this.#unprojectedModels.map(model => {
if (model.provider !== providerName) return model;
return this.#applyProviderTransportOverrideToModel(model, transportOverride);
@@ -1158,7 +1158,7 @@ describe("ModelRegistry runtime discovery", () => {
const plain = registry.find("llama.cpp", "llama-3.1-8b");
expect(plain?.reasoning).toBe(false);
expect(plain?.api).toBe("openai-responses");
expect(plain?.baseUrl).toBe("http://127.0.0.1:8080");
expect(plain?.baseUrl).toBe("http://127.0.0.1:8080/v1");
expect((plain?.compat as DialectFields | undefined)?.reasoningDisableMode).not.toBe("qwen-template-false");
});