diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 0c73aa111..72dabb0b5 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747)) + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index a33fe6cf9..c1e2a046b 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3032,6 +3032,11 @@ export interface FetchLiteLLMRichModelsOptions { } type LiteLLMRichModelEntry = Record; +type LiteLLMRichEndpointModel = { + model: ModelSpec; + supportsVision: unknown; + supportsReasoning: unknown; +}; const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const; export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000; @@ -3264,7 +3269,7 @@ async function fetchLiteLLMRichEndpoint( managementBaseUrl: string, runtimeBaseUrl: string, signal?: AbortSignal, -): Promise[] | null> { +): Promise<{ models: LiteLLMRichEndpointModel[]; incompleteVisionMetadata: boolean } | null> { const fetchImpl = discoveryFetch(options.fetch); const requestHeaders: Record = { Accept: "application/json", @@ -3296,17 +3301,29 @@ async function fetchLiteLLMRichEndpoint( if (!entries || entries.length === 0) { return null; } - const deduped = new Map>(); + const deduped = new Map>(); + let incompleteVisionMetadata = false; for (const entry of entries) { const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl); if (model) { - deduped.set(model.id, model); + const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision"); + if (supportsVision !== true && supportsVision !== false) { + incompleteVisionMetadata = true; + } + deduped.set(model.id, { + model, + supportsVision, + supportsReasoning: getLiteLLMMetadataValue(entry, "supports_reasoning"), + }); } } if (deduped.size === 0) { return null; } - return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id)); + return { + models: Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id)), + incompleteVisionMetadata, + }; } export async function fetchLiteLLMRichModels( @@ -3318,13 +3335,40 @@ export async function fetchLiteLLMRichModels( return null; } const fetchModels = async (signal?: AbortSignal): Promise[] | null> => { + const deduped = new Map>(); for (const endpoint of LITELLM_RICH_ENDPOINTS) { - const models = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); - if (models) { - return models; + const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); + if (!result) { + continue; + } + for (const next of result.models) { + const existing = deduped.get(next.model.id); + if (!existing) { + deduped.set(next.model.id, next); + continue; + } + const model = { + ...existing.model, + ...next.model, + name: next.model.name === next.model.id ? existing.model.name : next.model.name, + input: + next.supportsVision === true || next.supportsVision === false + ? next.model.input + : existing.model.input, + reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning, + }; + deduped.set(next.model.id, { ...next, model }); + } + if (!result.incompleteVisionMetadata) { + break; } } - return null; + if (deduped.size === 0) { + return null; + } + return Array.from(deduped.values()) + .map(entry => entry.model) + .sort((left, right) => left.id.localeCompare(right.id)); }; if (options.signal !== undefined) { return fetchModels(options.signal); @@ -3340,10 +3384,11 @@ export function litellmModelManagerOptions( return { providerId: "litellm", // rich-v4 invalidates rows cached before LiteLLM ids gained bundled - // reference fallback. Earlier versions also handled reseller usage-suffix - // stripping and placeholder-only `all-team-models` filtering; bump the - // version whenever the mappers below change, or warm authoritative caches - // keep serving pre-change rows for the full TTL. + // reference fallback and before discovery continued past `/model_group/info` + // when that endpoint omitted vision metadata. Earlier versions handled + // reseller usage-suffix stripping and placeholder-only `all-team-models` + // filtering; bump the version whenever the mappers below change, or warm + // authoritative caches keep serving pre-change rows for the full TTL. cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index 02e7f8664..2524d014d 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -295,8 +295,13 @@ describe("LiteLLM provider discovery", () => { if (url === "http://primary:4000/model_group/info") { return Response.json({ data: [ - { model_group: "no-tools", providers: ["openai"], supports_function_calling: false }, - { model_group: "params-tools", supported_openai_params: ["tools"] }, + { + model_group: "no-tools", + providers: ["openai"], + supports_vision: false, + supports_function_calling: false, + }, + { model_group: "params-tools", supports_vision: false, supported_openai_params: ["tools"] }, ], }); } @@ -387,6 +392,7 @@ describe("LiteLLM provider discovery", () => { max_input_tokens: 96_000, max_output_tokens: 8_000, supports_function_calling: true, + supports_vision: false, }, ], }); @@ -470,6 +476,70 @@ describe("LiteLLM provider discovery", () => { }); }); + test("continues to LiteLLM model info when model_group omits vision metadata", async () => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "vision-proxy-model", + model_name: "Vision Proxy Model", + max_input_tokens: 128_000, + max_output_tokens: 16_000, + }, + ], + }); + } + if (url === "http://primary:4000/v2/model/info") { + return new Response("{}", { status: 404 }); + } + if (url === "http://primary:4000/model/info") { + return Response.json({ + data: [ + { model_name: "text-only-model", model_info: { supports_vision: false } }, + { + model_name: "vision-proxy-model", + model_info: { + max_input_tokens: 128_000, + max_output_tokens: 16_000, + supports_vision: true, + }, + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when LiteLLM model info succeeds"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(calls).toContain("http://primary:4000/model_group/info"); + expect(calls).toContain("http://primary:4000/v2/model/info"); + expect(calls).toContain("http://primary:4000/model/info"); + expect(calls).not.toContain("http://primary:4000/v1/models"); + expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({ + id: "vision-proxy-model", + name: "Vision Proxy Model", + input: ["text", "image"], + contextWindow: 128_000, + maxTokens: 16_000, + }); + }); + test("falls back from v2 model info to LiteLLM model info", async () => { const calls: string[] = []; const fetchMock = vi.fn(async (input: string | URL | Request) => { @@ -479,7 +549,9 @@ describe("LiteLLM provider discovery", () => { return new Response("{}", { status: 404 }); } if (url === "http://primary:4000/model/info") { - return Response.json({ data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000 } }] }); + return Response.json({ + data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000, supports_vision: false } }], + }); } throw new Error(`Unexpected URL: ${url}`); }) as FetchImpl;