From a29acbfa408178507a956914e85df47995126536 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 4 Aug 2026 03:18:36 +0000 Subject: [PATCH] fix(config): read server input modalities in openai-models-list discovery discoverOpenAIModelsList never consulted the /v1/models row's input field, so custom virtual tier ids absent from the bundled catalog fell through to the ["text"] fallback and showed images: no even when the server advertised input: ["text","image"]. Parse the direct input array and OpenRouter-style architecture.input_modalities from the response row, preferring server-reported modalities over native metadata and the bundled reference. Fixes #7583 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/config/model-discovery.ts | 36 ++++++++++++++++- .../coding-agent/test/model-discovery.test.ts | 40 +++++++++++++++++++ 3 files changed, 78 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 84a743e43..0a3d85e2c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `openai-models-list` discovery ignoring server-advertised input modalities, so custom virtual tier IDs absent from the bundled catalog showed `images: no` even when the `/v1/models` response reported `input: ["text","image"]` ([#7583](https://github.com/can1357/oh-my-pi/issues/7583)). + ## [17.2.7] - 2026-08-03 ### Changed diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 7e30f0fd7..0e98ad377 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -724,6 +724,30 @@ export async function discoverLlamaCppModelRuntimeMetadata( } } +/** + * Read image-input support from an OpenAI-compatible `/v1/models` row. Handles + * the direct `input: ["text","image"]` array thin proxies emit and the + * OpenRouter-style `architecture.input_modalities` array; returns undefined when + * neither is present so the bundled reference (or the `["text"]` default) can + * take over. + */ +function extractOpenAIModelsListInputCapabilities(item: { + input?: unknown; + architecture?: unknown; +}): ("text" | "image")[] | undefined { + const modalities = new Set(); + const collect = (value: unknown): void => { + if (!Array.isArray(value)) return; + for (const entry of value) { + if (typeof entry === "string") modalities.add(entry.toLowerCase()); + } + }; + collect(item.input); + if (isRecord(item.architecture)) collect(item.architecture.input_modalities); + if (modalities.size === 0) return undefined; + return modalities.has("image") ? ["text", "image"] : ["text"]; +} + export async function discoverOpenAIModelsList( providerConfig: DiscoveryProviderConfig, ctx: DiscoveryContext, @@ -752,7 +776,13 @@ export async function discoverOpenAIModelsList( } headers = h; return (await res.json()) as { - data?: Array<{ id?: string; max_model_len?: unknown; context_length?: unknown }>; + data?: Array<{ + id?: string; + max_model_len?: unknown; + context_length?: unknown; + input?: unknown; + architecture?: unknown; + }>; }; }), nativeMetadataPromise, @@ -796,7 +826,9 @@ export async function discoverOpenAIModelsList( baseUrl, reasoning: reference?.reasoning ?? false, thinking: inheritReferenceThinking(undefined, reference, providerConfig.provider), - input: nativeMetadataForModel?.input ?? reference?.input ?? ["text"], + input: extractOpenAIModelsListInputCapabilities(item) ?? + nativeMetadataForModel?.input ?? + reference?.input ?? ["text"], ...(providerConfig.discovery.type === "lm-studio" ? { imageInputDecoder: "stb" as const } : {}), // Proxy/gateway pricing is provider-specific and rarely matches // upstream bundled catalogs, so keep costs local-unknown even diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index 2b740a3c7..15f1d2f19 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -2172,6 +2172,46 @@ providers: expect(unknown?.reasoning).toBe(false); }); + test("openai-models-list discovery reads server-advertised input modalities for ids absent from the catalog", async () => { + writeRawModelsJson({ + "openai-test": { + baseUrl: "http://127.0.0.1:9996", + api: "openai-completions", + auth: "none", + discovery: { type: "openai-models-list" }, + }, + }); + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:9996/v1/models") { + // Custom virtual tier ids that are absent from the bundled + // catalog: their vision support can only come from the server row. + return new Response( + JSON.stringify({ + data: [ + { id: "high", object: "model", input: ["text", "image"] }, + { id: "leftover", object: "model", architecture: { input_modalities: ["text", "image"] } }, + { id: "low", object: "model", input: ["text"] }, + { id: "medium", object: "model" }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); + await registry.refresh(); + // Direct `input` array and OpenRouter-style `architecture.input_modalities` + // both surface vision support. + expect(registry.find("openai-test", "high")?.input).toEqual(["text", "image"]); + expect(registry.find("openai-test", "leftover")?.input).toEqual(["text", "image"]); + // Server explicitly reports text-only; no image support invented. + expect(registry.find("openai-test", "low")?.input).toEqual(["text"]); + // Silent server → default text-only fallback. + expect(registry.find("openai-test", "medium")?.input).toEqual(["text"]); + }); + test("proxy discovery honors API-reported context_length and endpoint routing", async () => { writeRawModelsJson({ "proxy-test": {