merge PR #4749: fix(catalog): preserve LiteLLM vision metadata

This commit is contained in:
can1357
2026-07-08 15:22:20 +02:00
3 changed files with 136 additions and 15 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747))
## [16.3.11] - 2026-07-06
### Added
@@ -3032,6 +3032,11 @@ export interface FetchLiteLLMRichModelsOptions<TApi extends Api> {
}
type LiteLLMRichModelEntry = Record<string, unknown>;
type LiteLLMRichEndpointModel<TApi extends Api> = {
model: ModelSpec<TApi>;
supportsVision: unknown;
supportsReasoning: unknown;
};
const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const;
export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000;
@@ -3264,7 +3269,7 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
managementBaseUrl: string,
runtimeBaseUrl: string,
signal?: AbortSignal,
): Promise<ModelSpec<TApi>[] | null> {
): Promise<{ models: LiteLLMRichEndpointModel<TApi>[]; incompleteVisionMetadata: boolean } | null> {
const fetchImpl = discoveryFetch(options.fetch);
const requestHeaders: Record<string, string> = {
Accept: "application/json",
@@ -3296,17 +3301,29 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
if (!entries || entries.length === 0) {
return null;
}
const deduped = new Map<string, ModelSpec<TApi>>();
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
let incompleteVisionMetadata = false;
for (const entry of entries) {
const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl);
if (model) {
deduped.set(model.id, model);
const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision");
if (supportsVision !== true && supportsVision !== false) {
incompleteVisionMetadata = true;
}
deduped.set(model.id, {
model,
supportsVision,
supportsReasoning: getLiteLLMMetadataValue(entry, "supports_reasoning"),
});
}
}
if (deduped.size === 0) {
return null;
}
return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id));
return {
models: Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id)),
incompleteVisionMetadata,
};
}
export async function fetchLiteLLMRichModels<TApi extends Api>(
@@ -3318,13 +3335,40 @@ export async function fetchLiteLLMRichModels<TApi extends Api>(
return null;
}
const fetchModels = async (signal?: AbortSignal): Promise<ModelSpec<TApi>[] | null> => {
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
for (const endpoint of LITELLM_RICH_ENDPOINTS) {
const models = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
if (models) {
return models;
const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
if (!result) {
continue;
}
for (const next of result.models) {
const existing = deduped.get(next.model.id);
if (!existing) {
deduped.set(next.model.id, next);
continue;
}
const model = {
...existing.model,
...next.model,
name: next.model.name === next.model.id ? existing.model.name : next.model.name,
input:
next.supportsVision === true || next.supportsVision === false
? next.model.input
: existing.model.input,
reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning,
};
deduped.set(next.model.id, { ...next, model });
}
if (!result.incompleteVisionMetadata) {
break;
}
}
return null;
if (deduped.size === 0) {
return null;
}
return Array.from(deduped.values())
.map(entry => entry.model)
.sort((left, right) => left.id.localeCompare(right.id));
};
if (options.signal !== undefined) {
return fetchModels(options.signal);
@@ -3340,10 +3384,11 @@ export function litellmModelManagerOptions(
return {
providerId: "litellm",
// rich-v4 invalidates rows cached before LiteLLM ids gained bundled
// reference fallback. Earlier versions also handled reseller usage-suffix
// stripping and placeholder-only `all-team-models` filtering; bump the
// version whenever the mappers below change, or warm authoritative caches
// keep serving pre-change rows for the full TTL.
// reference fallback and before discovery continued past `/model_group/info`
// when that endpoint omitted vision metadata. Earlier versions handled
// reseller usage-suffix stripping and placeholder-only `all-team-models`
// filtering; bump the version whenever the mappers below change, or warm
// authoritative caches keep serving pre-change rows for the full TTL.
cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`,
// litellm is a local-only proxy and is never bundled in models.json (that
// would leak the machine's localhost catalog). Prefer the proxy's richer
+75 -3
View File
@@ -295,8 +295,13 @@ describe("LiteLLM provider discovery", () => {
if (url === "http://primary:4000/model_group/info") {
return Response.json({
data: [
{ model_group: "no-tools", providers: ["openai"], supports_function_calling: false },
{ model_group: "params-tools", supported_openai_params: ["tools"] },
{
model_group: "no-tools",
providers: ["openai"],
supports_vision: false,
supports_function_calling: false,
},
{ model_group: "params-tools", supports_vision: false, supported_openai_params: ["tools"] },
],
});
}
@@ -387,6 +392,7 @@ describe("LiteLLM provider discovery", () => {
max_input_tokens: 96_000,
max_output_tokens: 8_000,
supports_function_calling: true,
supports_vision: false,
},
],
});
@@ -470,6 +476,70 @@ describe("LiteLLM provider discovery", () => {
});
});
test("continues to LiteLLM model info when model_group omits vision metadata", async () => {
const calls: string[] = [];
const fetchMock = vi.fn(async (input: string | URL | Request) => {
const url = inputUrl(input);
calls.push(url);
if (url === MODELS_DEV_URL) {
return Response.json({});
}
if (url === "http://primary:4000/model_group/info") {
return Response.json({
data: [
{
model_group: "vision-proxy-model",
model_name: "Vision Proxy Model",
max_input_tokens: 128_000,
max_output_tokens: 16_000,
},
],
});
}
if (url === "http://primary:4000/v2/model/info") {
return new Response("{}", { status: 404 });
}
if (url === "http://primary:4000/model/info") {
return Response.json({
data: [
{ model_name: "text-only-model", model_info: { supports_vision: false } },
{
model_name: "vision-proxy-model",
model_info: {
max_input_tokens: 128_000,
max_output_tokens: 16_000,
supports_vision: true,
},
},
],
});
}
if (url === "http://primary:4000/v1/models") {
throw new Error("/v1/models should not be called when LiteLLM model info succeeds");
}
throw new Error(`Unexpected URL: ${url}`);
}) as FetchImpl;
const options = litellmModelManagerOptions({
apiKey: "sk-rich",
baseUrl: "http://primary:4000/v1",
fetch: fetchMock,
});
const models = await options.fetchDynamicModels?.();
expect(calls).toContain("http://primary:4000/model_group/info");
expect(calls).toContain("http://primary:4000/v2/model/info");
expect(calls).toContain("http://primary:4000/model/info");
expect(calls).not.toContain("http://primary:4000/v1/models");
expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({
id: "vision-proxy-model",
name: "Vision Proxy Model",
input: ["text", "image"],
contextWindow: 128_000,
maxTokens: 16_000,
});
});
test("falls back from v2 model info to LiteLLM model info", async () => {
const calls: string[] = [];
const fetchMock = vi.fn(async (input: string | URL | Request) => {
@@ -479,7 +549,9 @@ describe("LiteLLM provider discovery", () => {
return new Response("{}", { status: 404 });
}
if (url === "http://primary:4000/model/info") {
return Response.json({ data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000 } }] });
return Response.json({
data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000, supports_vision: false } }],
});
}
throw new Error(`Unexpected URL: ${url}`);
}) as FetchImpl;