From 15c7429ad07b7b2a910fff2f127cec13f43a4fdd Mon Sep 17 00:00:00 2001 From: Gregor <1044627+tarcon@users.noreply.github.com> Date: Fri, 13 Mar 2026 14:16:53 +0000 Subject: [PATCH] add llama.cpp as local provider (#370) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * add llama.cpp as local provider * use responses api instead of messages * use api-keys correctly for llama.cpp provider --------- Co-authored-by: Can Bölük --- README.md | 12 +- docs/environment-variables.md | 116 +++++++++--------- docs/models.md | 19 +++ packages/ai/CHANGELOG.md | 1 + packages/ai/README.md | 1 + packages/ai/src/stream.ts | 1 + packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/config/model-registry.ts | 115 ++++++++++++++--- .../coding-agent/test/model-registry.test.ts | 58 +++++++++ 9 files changed, 249 insertions(+), 75 deletions(-) diff --git a/README.md b/README.md index c573f4656..f4d188a71 100644 --- a/README.md +++ b/README.md @@ -476,7 +476,7 @@ return config **Option 1: Environment variables** (common examples) | Provider | Environment Variable | -| ----------------------------------------------- | -------------------------------------------- | +|-------------------------------------------------| -------------------------------------------- | | Anthropic | `ANTHROPIC_API_KEY` | | OpenAI | `OPENAI_API_KEY` | | Google | `GEMINI_API_KEY` | @@ -491,6 +491,7 @@ return config | Ollama (`ollama`) | `OLLAMA_API_KEY` _(optional)_ | | LiteLLM (`litellm`) | `LITELLM_API_KEY` | | LM Studio (`lm-studio`) | `LM_STUDIO_API_KEY` _(optional)_ | +| llama.cpp (`llama.cpp`) | `LLAMA_CPP_API_KEY` _(optional)_ | | Xiaomi MiMo (`xiaomi`) | `XIAOMI_API_KEY` | | Moonshot (`moonshot`) | `MOONSHOT_API_KEY` | | Venice (`venice`) | `VENICE_API_KEY` | @@ -529,6 +530,7 @@ Use `/login` with supported providers: - Qianfan (`qianfan`) - Ollama (local / self-hosted, `ollama`) - LM Studio (local / self-hosted, `lm-studio`) +- llama.cpp (local / self-hosted, `llama.cpp`) - vLLM (local OpenAI-compatible, `vllm`) - Z.AI (GLM Coding Plan) - Synthetic @@ -542,6 +544,7 @@ Use `/login` with supported providers: - Cloudflare AI Gateway (`cloudflare-ai-gateway`) For `ollama`, API key is optional. Leave it unset for local no-auth instances, or set `OLLAMA_API_KEY` for authenticated hosts. +For `llama.cpp`, API key is optional. Leave it unset for local no-auth instances, or set `LLAMA_CPP_API_KEY` for authenticated hosts. For `lm-studio`, API key is optional. Leave it unset for local no-auth instances, or set `LM_STUDIO_API_KEY` for authenticated hosts. For `vllm`, paste your key in `/login` (or use `VLLM_API_KEY`). For local no-auth servers, any placeholder value works (for example `vllm-local`). For `nanogpt`, `/login nanogpt` opens `https://nano-gpt.com/api` and prompts for your `sk-...` key (or set `NANO_GPT_API_KEY`). Login validates the key via NanoGPT's models endpoint (not a fixed model entitlement). @@ -854,6 +857,13 @@ providers: cacheWrite: 0 contextWindow: 128000 maxTokens: 32000 + + llama.cpp: + baseUrl: http://127.0.0.1:8080 + api: openai-responses + auth: none + discovery: + type: llama.cpp ``` **Supported APIs:** `openai-completions`, `openai-responses`, `openai-codex-responses`, `azure-openai-responses`, `anthropic-messages`, `google-generative-ai`, `google-vertex` diff --git a/docs/environment-variables.md b/docs/environment-variables.md index fef5df3a9..222a44ac5 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -28,44 +28,45 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not ### Core provider credentials -| Variable | Used for | Required when | Notes / precedence | -|---|---|---|---| -| `ANTHROPIC_OAUTH_TOKEN` | Anthropic API auth | Using Anthropic with OAuth token auth | Takes precedence over `ANTHROPIC_API_KEY` for provider auth resolution | -| `ANTHROPIC_API_KEY` | Anthropic API auth | Using Anthropic without OAuth token | Fallback after `ANTHROPIC_OAUTH_TOKEN` | -| `ANTHROPIC_FOUNDRY_API_KEY` | Anthropic via Azure Foundry / enterprise gateway | `CLAUDE_CODE_USE_FOUNDRY` enabled | Takes precedence over `ANTHROPIC_OAUTH_TOKEN` and `ANTHROPIC_API_KEY` when Foundry mode is enabled | -| `OPENAI_API_KEY` | OpenAI auth | Using OpenAI-family providers without explicit apiKey argument | Used by OpenAI Completions/Responses providers | -| `GEMINI_API_KEY` | Google Gemini auth | Using `google` provider models | Primary key for Gemini provider mapping | -| `GOOGLE_API_KEY` | Gemini image tool auth fallback | Using `gemini_image` tool without `GEMINI_API_KEY` | Used by coding-agent image tool fallback path | -| `GROQ_API_KEY` | Groq auth | Using Groq models | | -| `CEREBRAS_API_KEY` | Cerebras auth | Using Cerebras models | | -| `TOGETHER_API_KEY` | Together auth | Using `together` provider | | -| `HUGGINGFACE_HUB_TOKEN` | Hugging Face auth | Using `huggingface` provider | Primary Hugging Face token env var | -| `HF_TOKEN` | Hugging Face auth | Using `huggingface` provider | Fallback when `HUGGINGFACE_HUB_TOKEN` is unset | -| `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | | -| `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | | -| `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | | -| `VENICE_API_KEY` | Venice auth | Using `venice` provider | | -| `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key | -| `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required | -| `OLLAMA_API_KEY` | Ollama auth (optional) | Using `ollama` provider with authenticated hosts | Local Ollama usually runs without auth; any non-empty token works when a key is required | -| `XIAOMI_API_KEY` | Xiaomi MiMo auth | Using `xiaomi` provider | | -| `MOONSHOT_API_KEY` | Moonshot auth | Using `moonshot` provider | | -| `XAI_API_KEY` | xAI auth | Using xAI models | | -| `OPENROUTER_API_KEY` | OpenRouter auth | Using OpenRouter models | Also used by image tool when preferred/auto provider is OpenRouter | -| `MISTRAL_API_KEY` | Mistral auth | Using Mistral models | | -| `ZAI_API_KEY` | z.ai auth | Using z.ai models | Also used by z.ai web search provider | -| `MINIMAX_API_KEY` | MiniMax auth | Using `minimax` provider | | -| `MINIMAX_CODE_API_KEY` | MiniMax Code auth | Using `minimax-code` provider | | -| `MINIMAX_CODE_CN_API_KEY` | MiniMax Code CN auth | Using `minimax-code-cn` provider | | -| `OPENCODE_API_KEY` | OpenCode auth | Using OpenCode models | | -| `QIANFAN_API_KEY` | Qianfan auth | Using `qianfan` provider | | -| `QWEN_OAUTH_TOKEN` | Qwen Portal auth | Using `qwen-portal` with OAuth token | Takes precedence over `QWEN_PORTAL_API_KEY` | -| `QWEN_PORTAL_API_KEY` | Qwen Portal auth | Using `qwen-portal` with API key | Fallback after `QWEN_OAUTH_TOKEN` | -| `ZENMUX_API_KEY` | ZenMux auth | Using `zenmux` provider | Used for ZenMux OpenAI and Anthropic-compatible routes | -| `VLLM_API_KEY` | vLLM auth/discovery opt-in | Using `vllm` provider (local OpenAI-compatible servers) | Any non-empty value works for no-auth local servers | -| `CURSOR_ACCESS_TOKEN` | Cursor provider auth | Using Cursor provider | | -| `AI_GATEWAY_API_KEY` | Vercel AI Gateway auth | Using `vercel-ai-gateway` provider | | -| `CLOUDFLARE_AI_GATEWAY_API_KEY` | Cloudflare AI Gateway auth | Using `cloudflare-ai-gateway` provider | Base URL must be configured as `https://gateway.ai.cloudflare.com/v1///anthropic` | +| Variable | Used for | Required when | Notes / precedence | +|---------------------------------|---|---------------------------------------------------------------|-----------------------------------------------------------------------------------------------------| +| `ANTHROPIC_OAUTH_TOKEN` | Anthropic API auth | Using Anthropic with OAuth token auth | Takes precedence over `ANTHROPIC_API_KEY` for provider auth resolution | +| `ANTHROPIC_API_KEY` | Anthropic API auth | Using Anthropic without OAuth token | Fallback after `ANTHROPIC_OAUTH_TOKEN` | +| `ANTHROPIC_FOUNDRY_API_KEY` | Anthropic via Azure Foundry / enterprise gateway | `CLAUDE_CODE_USE_FOUNDRY` enabled | Takes precedence over `ANTHROPIC_OAUTH_TOKEN` and `ANTHROPIC_API_KEY` when Foundry mode is enabled | +| `OPENAI_API_KEY` | OpenAI auth | Using OpenAI-family providers without explicit apiKey argument | Used by OpenAI Completions/Responses providers | +| `GEMINI_API_KEY` | Google Gemini auth | Using `google` provider models | Primary key for Gemini provider mapping | +| `GOOGLE_API_KEY` | Gemini image tool auth fallback | Using `gemini_image` tool without `GEMINI_API_KEY` | Used by coding-agent image tool fallback path | +| `GROQ_API_KEY` | Groq auth | Using Groq models | | +| `CEREBRAS_API_KEY` | Cerebras auth | Using Cerebras models | | +| `TOGETHER_API_KEY` | Together auth | Using `together` provider | | +| `HUGGINGFACE_HUB_TOKEN` | Hugging Face auth | Using `huggingface` provider | Primary Hugging Face token env var | +| `HF_TOKEN` | Hugging Face auth | Using `huggingface` provider | Fallback when `HUGGINGFACE_HUB_TOKEN` is unset | +| `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | | +| `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | | +| `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | | +| `VENICE_API_KEY` | Venice auth | Using `venice` provider | | +| `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key | +| `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required | +| `OLLAMA_API_KEY` | Ollama auth (optional) | Using `ollama` provider with authenticated hosts | Local Ollama usually runs without auth; any non-empty token works when a key is required | +| `LLAMA_CPP_API_KEY` | Ollama auth (optional) | Using `llama-server` with `--api-key` parameter | Local llama.cpp usually runs without auth; any non-empty token works when a key is configured | +| `XIAOMI_API_KEY` | Xiaomi MiMo auth | Using `xiaomi` provider | | +| `MOONSHOT_API_KEY` | Moonshot auth | Using `moonshot` provider | | +| `XAI_API_KEY` | xAI auth | Using xAI models | | +| `OPENROUTER_API_KEY` | OpenRouter auth | Using OpenRouter models | Also used by image tool when preferred/auto provider is OpenRouter | +| `MISTRAL_API_KEY` | Mistral auth | Using Mistral models | | +| `ZAI_API_KEY` | z.ai auth | Using z.ai models | Also used by z.ai web search provider | +| `MINIMAX_API_KEY` | MiniMax auth | Using `minimax` provider | | +| `MINIMAX_CODE_API_KEY` | MiniMax Code auth | Using `minimax-code` provider | | +| `MINIMAX_CODE_CN_API_KEY` | MiniMax Code CN auth | Using `minimax-code-cn` provider | | +| `OPENCODE_API_KEY` | OpenCode auth | Using OpenCode models | | +| `QIANFAN_API_KEY` | Qianfan auth | Using `qianfan` provider | | +| `QWEN_OAUTH_TOKEN` | Qwen Portal auth | Using `qwen-portal` with OAuth token | Takes precedence over `QWEN_PORTAL_API_KEY` | +| `QWEN_PORTAL_API_KEY` | Qwen Portal auth | Using `qwen-portal` with API key | Fallback after `QWEN_OAUTH_TOKEN` | +| `ZENMUX_API_KEY` | ZenMux auth | Using `zenmux` provider | Used for ZenMux OpenAI and Anthropic-compatible routes | +| `VLLM_API_KEY` | vLLM auth/discovery opt-in | Using `vllm` provider (local OpenAI-compatible servers) | Any non-empty value works for no-auth local servers | +| `CURSOR_ACCESS_TOKEN` | Cursor provider auth | Using Cursor provider | | +| `AI_GATEWAY_API_KEY` | Vercel AI Gateway auth | Using `vercel-ai-gateway` provider | | +| `CLOUDFLARE_AI_GATEWAY_API_KEY` | Cloudflare AI Gateway auth | Using `cloudflare-ai-gateway` provider | Base URL must be configured as `https://gateway.ai.cloudflare.com/v1///anthropic` | ### GitHub/Copilot token chains @@ -241,25 +242,26 @@ Extra conditional behavior: ## 5) Agent/runtime behavior toggles -| Variable | Default / behavior | -|---|---| -| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) | -| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) | -| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) | -| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message | -| `NULL_PROMPT` | If `true`, system prompt builder returns empty string | -| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool | -| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) | -| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) | -| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) | -| `PI_TIMING` | If `1`, enables startup/tool timing instrumentation logs | -| `PI_DEBUG_STARTUP` | Enables startup stage debug prints to stderr in multiple startup paths | -| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (docs/examples/changelog path lookup) | -| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning | -| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) | -| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) | -| `PI_EDIT_VARIANT` | If `hashline`, forces hashline read/grep display mode when edit tool available | -| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool | +| Variable | Default / behavior | +|----------------------------|----------------------------------------------------------------------------------------------| +| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) | +| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) | +| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) | +| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message | +| `NULL_PROMPT` | If `true`, system prompt builder returns empty string | +| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool | +| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) | +| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) | +| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) | +| `PI_TIMING` | If `1`, enables startup/tool timing instrumentation logs | +| `PI_DEBUG_STARTUP` | Enables startup stage debug prints to stderr in multiple startup paths | +| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (docs/examples/changelog path lookup) | +| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning | +| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) | +| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) | +| `LLAMA_CPP_BASE_URL` | Default implicit Llama.cpp discovery base URL override (`http://127.0.0.1:8080` if unset) | +| `PI_EDIT_VARIANT` | If `hashline`, forces hashline read/grep display mode when edit tool available | +| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool | `PI_NO_PTY` is also set internally when CLI `--no-pty` is used. diff --git a/docs/models.md b/docs/models.md index c0f5b3879..50e4e87bd 100644 --- a/docs/models.md +++ b/docs/models.md @@ -151,6 +151,18 @@ If `ollama` is not explicitly configured, registry adds an implicit discoverable Runtime discovery calls `GET /api/tags` on Ollama and synthesizes model entries with local defaults. +### Implicit llama.cpp discovery + +If `llama.cpp` is not explicitly configured, registry adds an implicit discoverable provider: +Note: it's using the newer antropic messages api instead of the openai-competions. + +- provider: `llama.cpp` +- api: `openai-responses` +- base URL: `LLAMA_CPP_BASE_URL` or `http://127.0.0.1:8080` +- auth mode: keyless (`auth: none` behavior) + +Runtime discovery calls `GET models` on llama.cpp and synthesizes model entries with local defaults. + ### Implicit LM Studio discovery If `lm-studio` is not explicitly configured, registry adds an implicit discoverable provider: @@ -174,6 +186,13 @@ providers: auth: none discovery: type: ollama + + llama.cpp: + baseUrl: http://127.0.0.1:8080 + api: openai-responses + auth: none + discovery: + type: llama.cpp ``` ### Extension provider registration diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9e11b485d..1f4dfc7da 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Added `llama.cpp` as local provider - Fixed auth schema V0-to-V1 migration crash when the V0 table lacks a `disabled` column ## [13.11.0] - 2026-03-12 diff --git a/packages/ai/README.md b/packages/ai/README.md index 3ba745216..9fd7d8651 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -72,6 +72,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an - **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`) - **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL) - **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`) +- **llama.cpp** (local OpenAI and Anthropic compatible inference server) - **vLLM** (OpenAI-compatible server; `VLLM_API_KEY` for secured deployments) - **GitHub Copilot** (requires OAuth, see below) - **Google Gemini CLI** (requires OAuth, see below) diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 5e3abfdd1..5fa93dbce 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -135,6 +135,7 @@ const serviceProviderMap: Record = { nanogpt: "NANO_GPT_API_KEY", "lm-studio": "LM_STUDIO_API_KEY", ollama: "OLLAMA_API_KEY", + "llama.cpp": "LLAMA_CPP_API_KEY", qianfan: "QIANFAN_API_KEY", "qwen-portal": () => $pickenv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"), together: "TOGETHER_API_KEY", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 73a22c16a..00af356a5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Added +- Added `llama.cpp` as local provider - Added `code_search` tool supporting both Exa and grep.app providers for code snippet and documentation search - Added `providers.codeSearch` setting to configure code search provider (exa or grep) - Added grep.app integration for public code search with result ranking by context relevance diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 7f2ffaafa..9a1a5a729 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -160,7 +160,7 @@ const ModelOverrideSchema = Type.Object({ type ModelOverride = Static; const ProviderDiscoverySchema = Type.Object({ - type: Type.Union([Type.Literal("ollama"), Type.Literal("lm-studio")]), + type: Type.Union([Type.Literal("ollama"), Type.Literal("llama.cpp"), Type.Literal("lm-studio")]), }); const ProviderAuthSchema = Type.Union([Type.Literal("apiKey"), Type.Literal("none")]); @@ -694,6 +694,19 @@ export class ModelRegistry { }); this.#keylessProviders.add("ollama"); } + if (!configuredProviders.has("llama.cpp")) { + this.#discoverableProviders.push({ + provider: "llama.cpp", + api: "openai-responses", + baseUrl: Bun.env.LLAMA_CPP_BASE_URL || "http://127.0.0.1:8080", + discovery: { type: "llama.cpp" }, + optional: true, + }); + // Only mark as keyless if no API key is configured + if (!this.authStorage.hasAuth("llama.cpp")) { + this.#keylessProviders.add("llama.cpp"); + } + } if (!configuredProviders.has("lm-studio")) { this.#discoverableProviders.push({ provider: "lm-studio", @@ -851,30 +864,28 @@ export class ModelRegistry { } } - let fetchError: string | undefined; + const providerId = providerConfig.provider; + let discoveryError: string | undefined; const fetchDynamicModels = async (): Promise[] | null> => { try { - const models = - providerConfig.discovery.type === "ollama" - ? await this.#discoverOllamaModels(providerConfig) - : await this.#discoverLmStudioModels(providerConfig); - this.#lastDiscoveryWarnings.delete(providerConfig.provider); + const models = await this.#discoverModelsByProviderType(providerConfig); + this.#lastDiscoveryWarnings.delete(providerId); return models; } catch (error) { - fetchError = error instanceof Error ? error.message : String(error); + discoveryError = error instanceof Error ? error.message : String(error); return null; } }; const manager = createModelManager({ - providerId: providerConfig.provider, + providerId, staticModels: [], cacheDbPath: this.#cacheDbPath, cacheTtlMs: 24 * 60 * 60 * 1000, fetchDynamicModels, }); const result = await manager.refresh(strategy); - const status = fetchError + const status = discoveryError ? result.models.length > 0 ? "cached" : "unavailable" @@ -883,19 +894,30 @@ export class ModelRegistry { : cached ? "cached" : "idle"; - this.#providerDiscoveryStates.set(providerConfig.provider, { - provider: providerConfig.provider, + this.#providerDiscoveryStates.set(providerId, { + provider: providerId, status, optional: providerConfig.optional ?? false, stale: result.stale || status === "cached", - fetchedAt: fetchError ? cached?.updatedAt : Date.now(), + fetchedAt: discoveryError ? cached?.updatedAt : Date.now(), models: result.models.map(model => model.id), - error: fetchError, + error: discoveryError, }); - if (fetchError) { - this.#warnProviderDiscoveryFailure(providerConfig, fetchError); + if (discoveryError) { + this.#warnProviderDiscoveryFailure(providerConfig, discoveryError); + } + return this.#applyProviderModelOverrides(providerId, result.models); + } + + #discoverModelsByProviderType(providerConfig: DiscoveryProviderConfig): Promise[]> { + switch (providerConfig.discovery.type) { + case "ollama": + return this.#discoverOllamaModels(providerConfig); + case "llama.cpp": + return this.#discoverLlamaCppModels(providerConfig); + case "lm-studio": + return this.#discoverLmStudioModels(providerConfig); } - return this.#applyProviderModelOverrides(providerConfig.provider, result.models); } #warnProviderDiscoveryFailure(providerConfig: DiscoveryProviderConfig, error: string): void { @@ -1106,6 +1128,53 @@ export class ModelRegistry { return this.#applyProviderModelOverrides(providerConfig.provider, discovered); } + async #discoverLlamaCppModels(providerConfig: DiscoveryProviderConfig): Promise[]> { + const baseUrl = this.#normalizeLlamaCppBaseUrl(providerConfig.baseUrl); + const modelsUrl = `${baseUrl}/models`; + + const headers: Record = { ...(providerConfig.headers ?? {}) }; + const apiKey = await this.authStorage.getApiKey(providerConfig.provider); + if (apiKey && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== kNoAuth) { + headers.Authorization = `Bearer ${apiKey}`; + } + + const response = await fetch(modelsUrl, { + headers, + signal: AbortSignal.timeout(250), + }); + if (!response.ok) { + throw new Error(`HTTP ${response.status} from ${modelsUrl}`); + } + const payload = (await response.json()) as { data?: Array<{ id: string }> }; + const models = payload.data ?? []; + const discovered: Model[] = []; + for (const item of models) { + const id = item.id; + if (!id) continue; + discovered.push( + enrichModelThinking({ + id, + name: id, + api: providerConfig.api, + provider: providerConfig.provider, + baseUrl, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128000, + maxTokens: 8192, + headers, + compat: { + supportsStore: false, + supportsDeveloperRole: false, + supportsReasoningEffort: false, + }, + }), + ); + } + return this.#applyProviderModelOverrides(providerConfig.provider, discovered); + } + async #discoverLmStudioModels(providerConfig: DiscoveryProviderConfig): Promise[]> { const baseUrl = this.#normalizeLmStudioBaseUrl(providerConfig.baseUrl); const modelsUrl = `${baseUrl}/models`; @@ -1153,6 +1222,18 @@ export class ModelRegistry { return this.#applyProviderModelOverrides(providerConfig.provider, discovered); } + #normalizeLlamaCppBaseUrl(baseUrl?: string): string { + const defaultBaseUrl = "http://127.0.0.1:8080"; + const raw = baseUrl || defaultBaseUrl; + try { + const parsed = new URL(raw); + const trimmedPath = parsed.pathname.replace(/\/+$/g, ""); + return `${parsed.protocol}//${parsed.host}${trimmedPath}`; + } catch { + return raw; + } + } + #normalizeLmStudioBaseUrl(baseUrl?: string): string { const defaultBaseUrl = "http://127.0.0.1:1234/v1"; const raw = baseUrl || defaultBaseUrl; diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 59311d86d..a71237aed 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -905,5 +905,63 @@ describe("ModelRegistry", () => { expect(state?.status).toBe("unauthenticated"); expect(state?.models).toContain("local-coder"); }); + test("llama.cpp discovery honors configured API key", async () => { + authStorage.setRuntimeApiKey("llama.cpp", "test-llama-key"); + using _hook = hookFetch((input, init) => { + const url = String(input); + if (url === "http://127.0.0.1:8080/models") { + const headers = init?.headers as Headers | Record | undefined; + let authHeader: string | null = null; + if (headers instanceof Headers) { + authHeader = headers.get("Authorization"); + } else if (typeof headers === "object") { + authHeader = headers.Authorization; + } + expect(String(authHeader ?? "")).toBe("Bearer test-llama-key"); + return new Response(JSON.stringify({ data: [{ id: "llama-3.2:3b" }, { id: "mistral:7b" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + throw new Error(`Unexpected URL: ${url}`); + }); + const registry = new ModelRegistry(authStorage, modelsJsonPath); + await registry.refresh(); + const llamaModels = getModelsForProvider(registry, "llama.cpp"); + expect(llamaModels.some(m => m.id === "llama-3.2:3b")).toBe(true); + const apiKey = await registry.getApiKey(llamaModels[0]); + expect(apiKey).toBe("test-llama-key"); + expect(apiKey).not.toBe(kNoAuth); + }); + test("llama.cpp discovery without API key is treated as keyless", async () => { + using _hook = hookFetch((input, init) => { + const url = String(input); + if (url === "http://127.0.0.1:8080/models") { + const headers = init?.headers as Headers | Record | undefined; + let authHeader: string | null = null; + if (headers instanceof Headers) { + authHeader = headers.get("Authorization"); + } else if (typeof headers === "object") { + authHeader = headers.Authorization; + } + // When no API key, headers should be empty object or undefined + expect(authHeader).toBeUndefined(); + return new Response(JSON.stringify({ data: [{ id: "llama-3.2:3b" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + throw new Error(`Unexpected URL: ${url}`); + }); + const registry = new ModelRegistry(authStorage, modelsJsonPath); + await registry.refresh(); + const state = registry.getProviderDiscoveryState("llama.cpp"); + if (state?.status !== "ok") { + throw new Error(`Discovery failed with status ${state?.status}: ${state?.error}`); + } + const llamaModels = getModelsForProvider(registry, "llama.cpp"); + const apiKey = await registry.getApiKey(llamaModels[0]); + expect(apiKey).toBe(kNoAuth); + }); }); });