diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 328a7e2ae..3bfdeead2 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -323,6 +323,7 @@ Extra conditional behavior: | `SMITHERY_API_URL` | Smithery API base URL override (default `https://api.smithery.ai`) | | `SMITHERY_API_KEY` | Smithery API key for managed MCP auth lookup | | `PUPPETEER_EXECUTABLE_PATH` | Browser tool Chromium executable override | +| `LITELLM_BASE_URL` | LiteLLM proxy base URL fallback (`http://localhost:4000/v1` if unset); explicit `providers.litellm.baseUrl` / `models.yml` config wins | | `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) | | `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`OLLAMA_HOST` if unset, then `http://127.0.0.1:11434`) | | `OLLAMA_HOST` | Ollama host used for implicit Ollama discovery when `OLLAMA_BASE_URL` is unset; accepts Ollama-style values such as `127.0.0.1:11434` or `http://host:11434` | diff --git a/docs/models.md b/docs/models.md index 2286c3a47..3baf9b389 100644 --- a/docs/models.md +++ b/docs/models.md @@ -288,6 +288,17 @@ Runtime discovery fetches models (`GET /models`) and synthesizes model entries w This path also works for local OpenAI-compatible servers that are not LM Studio. For example, if oMLX is bound to Ollama's usual port, set `LM_STUDIO_BASE_URL=http://127.0.0.1:11434/v1` to discover it through the existing `/v1/models` flow. Running oMLX and Ollama side by side requires assigning a different port to one of them. Do not configure oMLX as `ollama`: Ollama discovery uses native `/api/tags` and `/api/show` endpoints, not OpenAI `/v1/models`. +### LiteLLM provider discovery + +When `litellm` is active (for example through `LITELLM_API_KEY` or stored auth), runtime discovery uses the LiteLLM proxy: + +- provider: `litellm` +- api: `openai-completions` +- base URL: explicit provider `baseUrl` / `models.yml` config, otherwise `LITELLM_BASE_URL`, otherwise `http://localhost:4000/v1` +- auth mode: `LITELLM_API_KEY` or stored LiteLLM auth when the proxy requires a key + +Runtime discovery fetches models (`GET /models`) from the proxy and enriches bare LiteLLM model ids against bundled reference metadata when available. + ### Explicit provider discovery You can configure discovery yourself: diff --git a/docs/providers.md b/docs/providers.md index 85c67c5ea..5629e8d53 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -109,7 +109,7 @@ Each provider has one or more environment variables that supply a key when no st | `venice` | `VENICE_API_KEY` | | `vercel-ai-gateway` | `AI_GATEWAY_API_KEY` (also `VERCEL_AI_GATEWAY_API_KEY` for catalog discovery) | | `cloudflare-ai-gateway` | `CLOUDFLARE_AI_GATEWAY_API_KEY` | -| `litellm` | `LITELLM_API_KEY` | +| `litellm` | `LITELLM_API_KEY`; optional `LITELLM_BASE_URL` for the proxy endpoint | | `kilo` | `KILO_API_KEY` | | `zai` | `ZAI_API_KEY` | | `zenmux` | `ZENMUX_API_KEY` | diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index b23030ad3..572e76905 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `LITELLM_BASE_URL` guidance to the LiteLLM login prompt so non-default proxy endpoints are discoverable. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726)) + ## [16.0.3] - 2026-06-16 ### Added diff --git a/packages/ai/src/registry/litellm.ts b/packages/ai/src/registry/litellm.ts index d32babfc6..6dea9c8ce 100644 --- a/packages/ai/src/registry/litellm.ts +++ b/packages/ai/src/registry/litellm.ts @@ -16,7 +16,8 @@ export async function loginLiteLLM(options: OAuthController): Promise { options.onAuth?.({ url: AUTH_URL, - instructions: "Run LiteLLM proxy (default http://localhost:4000/v1), then copy your master key or virtual key", + instructions: + "Run LiteLLM proxy (default http://localhost:4000/v1; set LITELLM_BASE_URL to customize it), then copy your master key or virtual key", }); const apiKey = await options.onPrompt({ diff --git a/packages/ai/test/litellm-login.test.ts b/packages/ai/test/litellm-login.test.ts new file mode 100644 index 000000000..4516fe215 --- /dev/null +++ b/packages/ai/test/litellm-login.test.ts @@ -0,0 +1,32 @@ +import { describe, expect, it } from "bun:test"; +import { loginLiteLLM } from "@oh-my-pi/pi-ai/registry/litellm"; + +describe("LiteLLM login", () => { + it("mentions LITELLM_BASE_URL for custom proxy endpoints", async () => { + let authInstructions: string | undefined; + let promptMessage: string | undefined; + + const apiKey = await loginLiteLLM({ + onAuth: info => { + authInstructions = info.instructions; + }, + onPrompt: async prompt => { + promptMessage = prompt.message; + return " sk-litellm-test "; + }, + }); + + expect(authInstructions).toContain("http://localhost:4000/v1"); + expect(authInstructions).toContain("LITELLM_BASE_URL"); + expect(promptMessage).toBe("Paste your LiteLLM API key (master key or virtual key)"); + expect(apiKey).toBe("sk-litellm-test"); + }); + + it("rejects empty keys", async () => { + await expect( + loginLiteLLM({ + onPrompt: async () => " ", + }), + ).rejects.toThrow("API key is required"); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 5c2321e0f..54a802f81 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `LITELLM_BASE_URL` as the LiteLLM provider discovery base URL fallback, with discovery caches scoped by the resolved proxy URL and explicit provider `baseUrl` config kept at higher precedence. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726)) + ## [16.0.2] - 2026-06-16 ### Fixed diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index a255e9970..7dbfde582 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2522,9 +2522,10 @@ export function litellmModelManagerOptions( config?: LiteLLMModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { const apiKey = config?.apiKey; - const baseUrl = config?.baseUrl ?? "http://localhost:4000/v1"; + const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1"; return { providerId: "litellm", + cacheProviderId: `litellm:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy whose /v1/models returns bare ids with no // metadata, and it is never bundled in models.json (that would leak the // machine's localhost catalog). It proxies known upstream models, so we diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts new file mode 100644 index 000000000..56799c0a1 --- /dev/null +++ b/packages/catalog/test/litellm-provider.test.ts @@ -0,0 +1,86 @@ +import { afterEach, describe, expect, test, vi } from "bun:test"; +import { litellmModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +const ORIGINAL_LITELLM_BASE_URL = Bun.env.LITELLM_BASE_URL; +const MODELS_DEV_URL = "https://models.dev/api.json"; + +function restoreLiteLLMBaseUrl(): void { + if (ORIGINAL_LITELLM_BASE_URL === undefined) { + delete Bun.env.LITELLM_BASE_URL; + return; + } + Bun.env.LITELLM_BASE_URL = ORIGINAL_LITELLM_BASE_URL; +} + +function inputUrl(input: string | URL | Request): string { + if (typeof input === "string") return input; + if (input instanceof URL) return input.toString(); + return input.url; +} + +function makeFetchMock(expectedModelUrl: string): FetchImpl { + return vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const url = inputUrl(input); + if (url === MODELS_DEV_URL) { + return new Response("{}", { status: 500 }); + } + + expect(url).toBe(expectedModelUrl); + expect(init?.method).toBe("GET"); + expect(init?.headers).toMatchObject({ + Accept: "application/json", + Authorization: "Bearer sk-litellm-test", + }); + return new Response(JSON.stringify({ data: [{ id: "openai/gpt-5" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }) as FetchImpl; +} + +afterEach(() => { + restoreLiteLLMBaseUrl(); + vi.restoreAllMocks(); +}); + +describe("LiteLLM provider discovery", () => { + test("uses LITELLM_BASE_URL when no explicit baseUrl is configured", async () => { + Bun.env.LITELLM_BASE_URL = "http://litellm.example:4100/v1"; + const fetchMock = makeFetchMock("http://litellm.example:4100/v1/models"); + + const options = litellmModelManagerOptions({ + apiKey: "sk-litellm-test", + fetch: fetchMock, + }); + const models = await options.fetchDynamicModels?.(); + + expect(options.cacheProviderId).toBe(`litellm:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`); + expect(fetchMock).toHaveBeenCalledTimes(2); + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "openai/gpt-5", + provider: "litellm", + baseUrl: "http://litellm.example:4100/v1", + }); + }); + + test("keeps explicit baseUrl higher precedence than LITELLM_BASE_URL", async () => { + Bun.env.LITELLM_BASE_URL = "http://litellm-env.example:4100/v1"; + const fetchMock = makeFetchMock("http://litellm-config.example:4200/v1/models"); + + const options = litellmModelManagerOptions({ + apiKey: "sk-litellm-test", + baseUrl: "http://litellm-config.example:4200/v1/", + fetch: fetchMock, + }); + const models = await options.fetchDynamicModels?.(); + + expect(options.cacheProviderId).toBe( + `litellm:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, + ); + expect(fetchMock).toHaveBeenCalledTimes(2); + expect(models).toHaveLength(1); + expect(models?.[0]?.baseUrl).toBe("http://litellm-config.example:4200/v1"); + }); +});