add llama.cpp as local provider (#370)

* add llama.cpp as local provider

* use responses api instead of messages

* use api-keys correctly for llama.cpp provider

---------

Co-authored-by: Can Bölük <can1357@users.noreply.github.com>
This commit is contained in:
Gregor
2026-03-13 14:16:53 +00:00
committed by GitHub
parent f6e373bb9e
commit 15c7429ad0
9 changed files with 249 additions and 75 deletions
+11 -1
View File
@@ -476,7 +476,7 @@ return config
**Option 1: Environment variables** (common examples)
| Provider | Environment Variable |
| ----------------------------------------------- | -------------------------------------------- |
|-------------------------------------------------| -------------------------------------------- |
| Anthropic | `ANTHROPIC_API_KEY` |
| OpenAI | `OPENAI_API_KEY` |
| Google | `GEMINI_API_KEY` |
@@ -491,6 +491,7 @@ return config
| Ollama (`ollama`) | `OLLAMA_API_KEY` _(optional)_ |
| LiteLLM (`litellm`) | `LITELLM_API_KEY` |
| LM Studio (`lm-studio`) | `LM_STUDIO_API_KEY` _(optional)_ |
| llama.cpp (`llama.cpp`) | `LLAMA_CPP_API_KEY` _(optional)_ |
| Xiaomi MiMo (`xiaomi`) | `XIAOMI_API_KEY` |
| Moonshot (`moonshot`) | `MOONSHOT_API_KEY` |
| Venice (`venice`) | `VENICE_API_KEY` |
@@ -529,6 +530,7 @@ Use `/login` with supported providers:
- Qianfan (`qianfan`)
- Ollama (local / self-hosted, `ollama`)
- LM Studio (local / self-hosted, `lm-studio`)
- llama.cpp (local / self-hosted, `llama.cpp`)
- vLLM (local OpenAI-compatible, `vllm`)
- Z.AI (GLM Coding Plan)
- Synthetic
@@ -542,6 +544,7 @@ Use `/login` with supported providers:
- Cloudflare AI Gateway (`cloudflare-ai-gateway`)
For `ollama`, API key is optional. Leave it unset for local no-auth instances, or set `OLLAMA_API_KEY` for authenticated hosts.
For `llama.cpp`, API key is optional. Leave it unset for local no-auth instances, or set `LLAMA_CPP_API_KEY` for authenticated hosts.
For `lm-studio`, API key is optional. Leave it unset for local no-auth instances, or set `LM_STUDIO_API_KEY` for authenticated hosts.
For `vllm`, paste your key in `/login` (or use `VLLM_API_KEY`). For local no-auth servers, any placeholder value works (for example `vllm-local`).
For `nanogpt`, `/login nanogpt` opens `https://nano-gpt.com/api` and prompts for your `sk-...` key (or set `NANO_GPT_API_KEY`). Login validates the key via NanoGPT's models endpoint (not a fixed model entitlement).
@@ -854,6 +857,13 @@ providers:
cacheWrite: 0
contextWindow: 128000
maxTokens: 32000
llama.cpp:
baseUrl: http://127.0.0.1:8080
api: openai-responses
auth: none
discovery:
type: llama.cpp
```
**Supported APIs:** `openai-completions`, `openai-responses`, `openai-codex-responses`, `azure-openai-responses`, `anthropic-messages`, `google-generative-ai`, `google-vertex`
+59 -57
View File
@@ -28,44 +28,45 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
### Core provider credentials
| Variable | Used for | Required when | Notes / precedence |
|---|---|---|---|
| `ANTHROPIC_OAUTH_TOKEN` | Anthropic API auth | Using Anthropic with OAuth token auth | Takes precedence over `ANTHROPIC_API_KEY` for provider auth resolution |
| `ANTHROPIC_API_KEY` | Anthropic API auth | Using Anthropic without OAuth token | Fallback after `ANTHROPIC_OAUTH_TOKEN` |
| `ANTHROPIC_FOUNDRY_API_KEY` | Anthropic via Azure Foundry / enterprise gateway | `CLAUDE_CODE_USE_FOUNDRY` enabled | Takes precedence over `ANTHROPIC_OAUTH_TOKEN` and `ANTHROPIC_API_KEY` when Foundry mode is enabled |
| `OPENAI_API_KEY` | OpenAI auth | Using OpenAI-family providers without explicit apiKey argument | Used by OpenAI Completions/Responses providers |
| `GEMINI_API_KEY` | Google Gemini auth | Using `google` provider models | Primary key for Gemini provider mapping |
| `GOOGLE_API_KEY` | Gemini image tool auth fallback | Using `gemini_image` tool without `GEMINI_API_KEY` | Used by coding-agent image tool fallback path |
| `GROQ_API_KEY` | Groq auth | Using Groq models | |
| `CEREBRAS_API_KEY` | Cerebras auth | Using Cerebras models | |
| `TOGETHER_API_KEY` | Together auth | Using `together` provider | |
| `HUGGINGFACE_HUB_TOKEN` | Hugging Face auth | Using `huggingface` provider | Primary Hugging Face token env var |
| `HF_TOKEN` | Hugging Face auth | Using `huggingface` provider | Fallback when `HUGGINGFACE_HUB_TOKEN` is unset |
| `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | |
| `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | |
| `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | |
| `VENICE_API_KEY` | Venice auth | Using `venice` provider | |
| `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key |
| `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required |
| `OLLAMA_API_KEY` | Ollama auth (optional) | Using `ollama` provider with authenticated hosts | Local Ollama usually runs without auth; any non-empty token works when a key is required |
| `XIAOMI_API_KEY` | Xiaomi MiMo auth | Using `xiaomi` provider | |
| `MOONSHOT_API_KEY` | Moonshot auth | Using `moonshot` provider | |
| `XAI_API_KEY` | xAI auth | Using xAI models | |
| `OPENROUTER_API_KEY` | OpenRouter auth | Using OpenRouter models | Also used by image tool when preferred/auto provider is OpenRouter |
| `MISTRAL_API_KEY` | Mistral auth | Using Mistral models | |
| `ZAI_API_KEY` | z.ai auth | Using z.ai models | Also used by z.ai web search provider |
| `MINIMAX_API_KEY` | MiniMax auth | Using `minimax` provider | |
| `MINIMAX_CODE_API_KEY` | MiniMax Code auth | Using `minimax-code` provider | |
| `MINIMAX_CODE_CN_API_KEY` | MiniMax Code CN auth | Using `minimax-code-cn` provider | |
| `OPENCODE_API_KEY` | OpenCode auth | Using OpenCode models | |
| `QIANFAN_API_KEY` | Qianfan auth | Using `qianfan` provider | |
| `QWEN_OAUTH_TOKEN` | Qwen Portal auth | Using `qwen-portal` with OAuth token | Takes precedence over `QWEN_PORTAL_API_KEY` |
| `QWEN_PORTAL_API_KEY` | Qwen Portal auth | Using `qwen-portal` with API key | Fallback after `QWEN_OAUTH_TOKEN` |
| `ZENMUX_API_KEY` | ZenMux auth | Using `zenmux` provider | Used for ZenMux OpenAI and Anthropic-compatible routes |
| `VLLM_API_KEY` | vLLM auth/discovery opt-in | Using `vllm` provider (local OpenAI-compatible servers) | Any non-empty value works for no-auth local servers |
| `CURSOR_ACCESS_TOKEN` | Cursor provider auth | Using Cursor provider | |
| `AI_GATEWAY_API_KEY` | Vercel AI Gateway auth | Using `vercel-ai-gateway` provider | |
| `CLOUDFLARE_AI_GATEWAY_API_KEY` | Cloudflare AI Gateway auth | Using `cloudflare-ai-gateway` provider | Base URL must be configured as `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic` |
| Variable | Used for | Required when | Notes / precedence |
|---------------------------------|---|---------------------------------------------------------------|-----------------------------------------------------------------------------------------------------|
| `ANTHROPIC_OAUTH_TOKEN` | Anthropic API auth | Using Anthropic with OAuth token auth | Takes precedence over `ANTHROPIC_API_KEY` for provider auth resolution |
| `ANTHROPIC_API_KEY` | Anthropic API auth | Using Anthropic without OAuth token | Fallback after `ANTHROPIC_OAUTH_TOKEN` |
| `ANTHROPIC_FOUNDRY_API_KEY` | Anthropic via Azure Foundry / enterprise gateway | `CLAUDE_CODE_USE_FOUNDRY` enabled | Takes precedence over `ANTHROPIC_OAUTH_TOKEN` and `ANTHROPIC_API_KEY` when Foundry mode is enabled |
| `OPENAI_API_KEY` | OpenAI auth | Using OpenAI-family providers without explicit apiKey argument | Used by OpenAI Completions/Responses providers |
| `GEMINI_API_KEY` | Google Gemini auth | Using `google` provider models | Primary key for Gemini provider mapping |
| `GOOGLE_API_KEY` | Gemini image tool auth fallback | Using `gemini_image` tool without `GEMINI_API_KEY` | Used by coding-agent image tool fallback path |
| `GROQ_API_KEY` | Groq auth | Using Groq models | |
| `CEREBRAS_API_KEY` | Cerebras auth | Using Cerebras models | |
| `TOGETHER_API_KEY` | Together auth | Using `together` provider | |
| `HUGGINGFACE_HUB_TOKEN` | Hugging Face auth | Using `huggingface` provider | Primary Hugging Face token env var |
| `HF_TOKEN` | Hugging Face auth | Using `huggingface` provider | Fallback when `HUGGINGFACE_HUB_TOKEN` is unset |
| `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | |
| `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | |
| `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | |
| `VENICE_API_KEY` | Venice auth | Using `venice` provider | |
| `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key |
| `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required |
| `OLLAMA_API_KEY` | Ollama auth (optional) | Using `ollama` provider with authenticated hosts | Local Ollama usually runs without auth; any non-empty token works when a key is required |
| `LLAMA_CPP_API_KEY` | Ollama auth (optional) | Using `llama-server` with `--api-key` parameter | Local llama.cpp usually runs without auth; any non-empty token works when a key is configured |
| `XIAOMI_API_KEY` | Xiaomi MiMo auth | Using `xiaomi` provider | |
| `MOONSHOT_API_KEY` | Moonshot auth | Using `moonshot` provider | |
| `XAI_API_KEY` | xAI auth | Using xAI models | |
| `OPENROUTER_API_KEY` | OpenRouter auth | Using OpenRouter models | Also used by image tool when preferred/auto provider is OpenRouter |
| `MISTRAL_API_KEY` | Mistral auth | Using Mistral models | |
| `ZAI_API_KEY` | z.ai auth | Using z.ai models | Also used by z.ai web search provider |
| `MINIMAX_API_KEY` | MiniMax auth | Using `minimax` provider | |
| `MINIMAX_CODE_API_KEY` | MiniMax Code auth | Using `minimax-code` provider | |
| `MINIMAX_CODE_CN_API_KEY` | MiniMax Code CN auth | Using `minimax-code-cn` provider | |
| `OPENCODE_API_KEY` | OpenCode auth | Using OpenCode models | |
| `QIANFAN_API_KEY` | Qianfan auth | Using `qianfan` provider | |
| `QWEN_OAUTH_TOKEN` | Qwen Portal auth | Using `qwen-portal` with OAuth token | Takes precedence over `QWEN_PORTAL_API_KEY` |
| `QWEN_PORTAL_API_KEY` | Qwen Portal auth | Using `qwen-portal` with API key | Fallback after `QWEN_OAUTH_TOKEN` |
| `ZENMUX_API_KEY` | ZenMux auth | Using `zenmux` provider | Used for ZenMux OpenAI and Anthropic-compatible routes |
| `VLLM_API_KEY` | vLLM auth/discovery opt-in | Using `vllm` provider (local OpenAI-compatible servers) | Any non-empty value works for no-auth local servers |
| `CURSOR_ACCESS_TOKEN` | Cursor provider auth | Using Cursor provider | |
| `AI_GATEWAY_API_KEY` | Vercel AI Gateway auth | Using `vercel-ai-gateway` provider | |
| `CLOUDFLARE_AI_GATEWAY_API_KEY` | Cloudflare AI Gateway auth | Using `cloudflare-ai-gateway` provider | Base URL must be configured as `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic` |
### GitHub/Copilot token chains
@@ -241,25 +242,26 @@ Extra conditional behavior:
## 5) Agent/runtime behavior toggles
| Variable | Default / behavior |
|---|---|
| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) |
| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) |
| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) |
| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message |
| `NULL_PROMPT` | If `true`, system prompt builder returns empty string |
| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool |
| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) |
| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) |
| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) |
| `PI_TIMING` | If `1`, enables startup/tool timing instrumentation logs |
| `PI_DEBUG_STARTUP` | Enables startup stage debug prints to stderr in multiple startup paths |
| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (docs/examples/changelog path lookup) |
| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning |
| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) |
| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) |
| `PI_EDIT_VARIANT` | If `hashline`, forces hashline read/grep display mode when edit tool available |
| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool |
| Variable | Default / behavior |
|----------------------------|----------------------------------------------------------------------------------------------|
| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) |
| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) |
| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) |
| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message |
| `NULL_PROMPT` | If `true`, system prompt builder returns empty string |
| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool |
| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) |
| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) |
| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) |
| `PI_TIMING` | If `1`, enables startup/tool timing instrumentation logs |
| `PI_DEBUG_STARTUP` | Enables startup stage debug prints to stderr in multiple startup paths |
| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (docs/examples/changelog path lookup) |
| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning |
| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) |
| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) |
| `LLAMA_CPP_BASE_URL` | Default implicit Llama.cpp discovery base URL override (`http://127.0.0.1:8080` if unset) |
| `PI_EDIT_VARIANT` | If `hashline`, forces hashline read/grep display mode when edit tool available |
| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool |
`PI_NO_PTY` is also set internally when CLI `--no-pty` is used.
+19
View File
@@ -151,6 +151,18 @@ If `ollama` is not explicitly configured, registry adds an implicit discoverable
Runtime discovery calls `GET /api/tags` on Ollama and synthesizes model entries with local defaults.
### Implicit llama.cpp discovery
If `llama.cpp` is not explicitly configured, registry adds an implicit discoverable provider:
Note: it's using the newer antropic messages api instead of the openai-competions.
- provider: `llama.cpp`
- api: `openai-responses`
- base URL: `LLAMA_CPP_BASE_URL` or `http://127.0.0.1:8080`
- auth mode: keyless (`auth: none` behavior)
Runtime discovery calls `GET models` on llama.cpp and synthesizes model entries with local defaults.
### Implicit LM Studio discovery
If `lm-studio` is not explicitly configured, registry adds an implicit discoverable provider:
@@ -174,6 +186,13 @@ providers:
auth: none
discovery:
type: ollama
llama.cpp:
baseUrl: http://127.0.0.1:8080
api: openai-responses
auth: none
discovery:
type: llama.cpp
```
### Extension provider registration
+1
View File
@@ -4,6 +4,7 @@
### Fixed
- Added `llama.cpp` as local provider
- Fixed auth schema V0-to-V1 migration crash when the V0 table lacks a `disabled` column
## [13.11.0] - 2026-03-12
+1
View File
@@ -72,6 +72,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
- **llama.cpp** (local OpenAI and Anthropic compatible inference server)
- **vLLM** (OpenAI-compatible server; `VLLM_API_KEY` for secured deployments)
- **GitHub Copilot** (requires OAuth, see below)
- **Google Gemini CLI** (requires OAuth, see below)
+1
View File
@@ -135,6 +135,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
nanogpt: "NANO_GPT_API_KEY",
"lm-studio": "LM_STUDIO_API_KEY",
ollama: "OLLAMA_API_KEY",
"llama.cpp": "LLAMA_CPP_API_KEY",
qianfan: "QIANFAN_API_KEY",
"qwen-portal": () => $pickenv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
together: "TOGETHER_API_KEY",
+1
View File
@@ -4,6 +4,7 @@
### Added
- Added `llama.cpp` as local provider
- Added `code_search` tool supporting both Exa and grep.app providers for code snippet and documentation search
- Added `providers.codeSearch` setting to configure code search provider (exa or grep)
- Added grep.app integration for public code search with result ranking by context relevance
@@ -160,7 +160,7 @@ const ModelOverrideSchema = Type.Object({
type ModelOverride = Static<typeof ModelOverrideSchema>;
const ProviderDiscoverySchema = Type.Object({
type: Type.Union([Type.Literal("ollama"), Type.Literal("lm-studio")]),
type: Type.Union([Type.Literal("ollama"), Type.Literal("llama.cpp"), Type.Literal("lm-studio")]),
});
const ProviderAuthSchema = Type.Union([Type.Literal("apiKey"), Type.Literal("none")]);
@@ -694,6 +694,19 @@ export class ModelRegistry {
});
this.#keylessProviders.add("ollama");
}
if (!configuredProviders.has("llama.cpp")) {
this.#discoverableProviders.push({
provider: "llama.cpp",
api: "openai-responses",
baseUrl: Bun.env.LLAMA_CPP_BASE_URL || "http://127.0.0.1:8080",
discovery: { type: "llama.cpp" },
optional: true,
});
// Only mark as keyless if no API key is configured
if (!this.authStorage.hasAuth("llama.cpp")) {
this.#keylessProviders.add("llama.cpp");
}
}
if (!configuredProviders.has("lm-studio")) {
this.#discoverableProviders.push({
provider: "lm-studio",
@@ -851,30 +864,28 @@ export class ModelRegistry {
}
}
let fetchError: string | undefined;
const providerId = providerConfig.provider;
let discoveryError: string | undefined;
const fetchDynamicModels = async (): Promise<readonly Model<Api>[] | null> => {
try {
const models =
providerConfig.discovery.type === "ollama"
? await this.#discoverOllamaModels(providerConfig)
: await this.#discoverLmStudioModels(providerConfig);
this.#lastDiscoveryWarnings.delete(providerConfig.provider);
const models = await this.#discoverModelsByProviderType(providerConfig);
this.#lastDiscoveryWarnings.delete(providerId);
return models;
} catch (error) {
fetchError = error instanceof Error ? error.message : String(error);
discoveryError = error instanceof Error ? error.message : String(error);
return null;
}
};
const manager = createModelManager<Api>({
providerId: providerConfig.provider,
providerId,
staticModels: [],
cacheDbPath: this.#cacheDbPath,
cacheTtlMs: 24 * 60 * 60 * 1000,
fetchDynamicModels,
});
const result = await manager.refresh(strategy);
const status = fetchError
const status = discoveryError
? result.models.length > 0
? "cached"
: "unavailable"
@@ -883,19 +894,30 @@ export class ModelRegistry {
: cached
? "cached"
: "idle";
this.#providerDiscoveryStates.set(providerConfig.provider, {
provider: providerConfig.provider,
this.#providerDiscoveryStates.set(providerId, {
provider: providerId,
status,
optional: providerConfig.optional ?? false,
stale: result.stale || status === "cached",
fetchedAt: fetchError ? cached?.updatedAt : Date.now(),
fetchedAt: discoveryError ? cached?.updatedAt : Date.now(),
models: result.models.map(model => model.id),
error: fetchError,
error: discoveryError,
});
if (fetchError) {
this.#warnProviderDiscoveryFailure(providerConfig, fetchError);
if (discoveryError) {
this.#warnProviderDiscoveryFailure(providerConfig, discoveryError);
}
return this.#applyProviderModelOverrides(providerId, result.models);
}
#discoverModelsByProviderType(providerConfig: DiscoveryProviderConfig): Promise<Model<Api>[]> {
switch (providerConfig.discovery.type) {
case "ollama":
return this.#discoverOllamaModels(providerConfig);
case "llama.cpp":
return this.#discoverLlamaCppModels(providerConfig);
case "lm-studio":
return this.#discoverLmStudioModels(providerConfig);
}
return this.#applyProviderModelOverrides(providerConfig.provider, result.models);
}
#warnProviderDiscoveryFailure(providerConfig: DiscoveryProviderConfig, error: string): void {
@@ -1106,6 +1128,53 @@ export class ModelRegistry {
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
}
async #discoverLlamaCppModels(providerConfig: DiscoveryProviderConfig): Promise<Model<Api>[]> {
const baseUrl = this.#normalizeLlamaCppBaseUrl(providerConfig.baseUrl);
const modelsUrl = `${baseUrl}/models`;
const headers: Record<string, string> = { ...(providerConfig.headers ?? {}) };
const apiKey = await this.authStorage.getApiKey(providerConfig.provider);
if (apiKey && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== kNoAuth) {
headers.Authorization = `Bearer ${apiKey}`;
}
const response = await fetch(modelsUrl, {
headers,
signal: AbortSignal.timeout(250),
});
if (!response.ok) {
throw new Error(`HTTP ${response.status} from ${modelsUrl}`);
}
const payload = (await response.json()) as { data?: Array<{ id: string }> };
const models = payload.data ?? [];
const discovered: Model<Api>[] = [];
for (const item of models) {
const id = item.id;
if (!id) continue;
discovered.push(
enrichModelThinking({
id,
name: id,
api: providerConfig.api,
provider: providerConfig.provider,
baseUrl,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 8192,
headers,
compat: {
supportsStore: false,
supportsDeveloperRole: false,
supportsReasoningEffort: false,
},
}),
);
}
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
}
async #discoverLmStudioModels(providerConfig: DiscoveryProviderConfig): Promise<Model<Api>[]> {
const baseUrl = this.#normalizeLmStudioBaseUrl(providerConfig.baseUrl);
const modelsUrl = `${baseUrl}/models`;
@@ -1153,6 +1222,18 @@ export class ModelRegistry {
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
}
#normalizeLlamaCppBaseUrl(baseUrl?: string): string {
const defaultBaseUrl = "http://127.0.0.1:8080";
const raw = baseUrl || defaultBaseUrl;
try {
const parsed = new URL(raw);
const trimmedPath = parsed.pathname.replace(/\/+$/g, "");
return `${parsed.protocol}//${parsed.host}${trimmedPath}`;
} catch {
return raw;
}
}
#normalizeLmStudioBaseUrl(baseUrl?: string): string {
const defaultBaseUrl = "http://127.0.0.1:1234/v1";
const raw = baseUrl || defaultBaseUrl;
@@ -905,5 +905,63 @@ describe("ModelRegistry", () => {
expect(state?.status).toBe("unauthenticated");
expect(state?.models).toContain("local-coder");
});
test("llama.cpp discovery honors configured API key", async () => {
authStorage.setRuntimeApiKey("llama.cpp", "test-llama-key");
using _hook = hookFetch((input, init) => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
const headers = init?.headers as Headers | Record<string, string> | undefined;
let authHeader: string | null = null;
if (headers instanceof Headers) {
authHeader = headers.get("Authorization");
} else if (typeof headers === "object") {
authHeader = headers.Authorization;
}
expect(String(authHeader ?? "")).toBe("Bearer test-llama-key");
return new Response(JSON.stringify({ data: [{ id: "llama-3.2:3b" }, { id: "mistral:7b" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
throw new Error(`Unexpected URL: ${url}`);
});
const registry = new ModelRegistry(authStorage, modelsJsonPath);
await registry.refresh();
const llamaModels = getModelsForProvider(registry, "llama.cpp");
expect(llamaModels.some(m => m.id === "llama-3.2:3b")).toBe(true);
const apiKey = await registry.getApiKey(llamaModels[0]);
expect(apiKey).toBe("test-llama-key");
expect(apiKey).not.toBe(kNoAuth);
});
test("llama.cpp discovery without API key is treated as keyless", async () => {
using _hook = hookFetch((input, init) => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
const headers = init?.headers as Headers | Record<string, string> | undefined;
let authHeader: string | null = null;
if (headers instanceof Headers) {
authHeader = headers.get("Authorization");
} else if (typeof headers === "object") {
authHeader = headers.Authorization;
}
// When no API key, headers should be empty object or undefined
expect(authHeader).toBeUndefined();
return new Response(JSON.stringify({ data: [{ id: "llama-3.2:3b" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
throw new Error(`Unexpected URL: ${url}`);
});
const registry = new ModelRegistry(authStorage, modelsJsonPath);
await registry.refresh();
const state = registry.getProviderDiscoveryState("llama.cpp");
if (state?.status !== "ok") {
throw new Error(`Discovery failed with status ${state?.status}: ${state?.error}`);
}
const llamaModels = getModelsForProvider(registry, "llama.cpp");
const apiKey = await registry.getApiKey(llamaModels[0]);
expect(apiKey).toBe(kNoAuth);
});
});
});