diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index b24291117..57cf4750c 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Umans `umans-glm-5.1` / `umans-glm-5.2` advertising native image input. The `models/info` endpoint reports `supports_vision: "via-handoff"` for the GLM models, meaning vision routes through a separate handoff pre-analysis step instead of accepting raw image blocks; `umansSupportsVision` treated any non-empty string as native vision support, so image prompts went directly to GLM and were rejected with `400 This model does not support image inputs`. The helper now requires `supports_vision === true`, the bundled GLM 5.1/5.2 rows are corrected to text-only, and stale mismatched Umans cache rows for those ids are dropped so the vision-handoff path runs even before a successful refresh. ([#3184](https://github.com/can1357/oh-my-pi/issues/3184)) + ## [16.1.9] - 2026-06-21 ### Fixed diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index e348d3274..ab69a18ff 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -39,6 +39,8 @@ export interface ModelManagerOptions Promise[] | null>; /** Optional models.dev fallback hook. */ @@ -148,7 +150,13 @@ export async function resolveProviderModels(cache?.models ?? []); + const cacheModels = dynamicFetchSucceeded + ? [] + : dropCachedModelIdsOnStaticMismatch( + normalizeModelList(cache?.models ?? []), + cacheFingerprintMatches, + options.dropCachedModelIdsOnStaticMismatch, + ); const dynamicModels = fetchedDynamicModels ?? []; const mergedWithCache = mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), cacheModels); const mergedModels = mergeDynamicModels(mergedWithCache, dynamicModels); @@ -180,7 +188,11 @@ export async function resolveProviderModels(latestCache?.models ?? cache?.models ?? []), + dropCachedModelIdsOnStaticMismatch( + normalizeModelList(latestCache?.models ?? cache?.models ?? []), + cacheFingerprintMatches, + options.dropCachedModelIdsOnStaticMismatch, + ), ), ), false, @@ -248,6 +260,18 @@ function shouldFetchRemoteSources( return false; } +function dropCachedModelIdsOnStaticMismatch( + models: readonly Model[], + cacheFingerprintMatches: boolean, + ids: readonly string[] | undefined, +): Model[] { + if (cacheFingerprintMatches || ids === undefined || ids.length === 0 || models.length === 0) { + return models.length === 0 ? [] : [...models]; + } + const droppedIds = new Set(ids); + return models.filter(model => !droppedIds.has(model.id)); +} + function mergeModelSources(...sources: readonly (readonly Model[])[]): Model[] { // Strip out empty/missing sources up front. The hot path is `(static, [])` // (modelsDev disabled / failed) — a single non-empty source means we can diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 7c18a7ebb..1acd7edf6 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -69284,8 +69284,7 @@ "baseUrl": "https://api.code.umans.ai", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -69327,8 +69326,7 @@ ] }, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 4f7e16ad7..90a6d8dd1 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -568,6 +568,7 @@ const UMANS_REASONING_EFFORT_BY_LEVEL: Record = { xhigh: Effort.XHigh, }; const UMANS_DEFAULT_REASONING_EFFORTS = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const; +const UMANS_VIA_HANDOFF_MODEL_IDS = ["umans-glm-5.1", "umans-glm-5.2"] as const; export interface UmansModelManagerConfig { apiKey?: string; @@ -586,8 +587,18 @@ function normalizeUmansBaseUrl(baseUrl: string | undefined): string { return normalized.endsWith("/v1") ? normalized.slice(0, -3) : normalized; } +/** + * Umans `models/info` reports `supports_vision: true` for natively + * vision-capable models and a non-empty string sentinel (e.g. + * `"via-handoff"`) for models that route image inputs through a vision + * handoff pre-analysis step instead of accepting raw image blocks. Only + * `true` means the model accepts image content directly; sentinel values + * MUST map to text-only so the agent's vision-handoff path runs instead + * of triggering an upstream HTTP 400 (`This model does not support image + * inputs`). + */ function umansSupportsVision(value: unknown): boolean { - return value === true || (typeof value === "string" && value.length > 0); + return value === true; } function umansReasoningSupported(value: unknown): boolean { @@ -704,6 +715,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode return { providerId: "umans", dynamicModelsAuthoritative: true, + dropCachedModelIdsOnStaticMismatch: UMANS_VIA_HANDOFF_MODEL_IDS, fetchDynamicModels: () => fetchUmansModelsInfo({ baseUrl, apiKey, fetch: config?.fetch, references }), }; } diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index e30bd3667..1b7eb0030 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -1,10 +1,14 @@ import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager"; import { MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, umansModelManagerOptions, } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; +import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; import modelsJson from "../src/models.json"; interface BundledModel { @@ -101,6 +105,116 @@ describe("umans provider catalog", () => { await expect(fetchDynamicModels()).rejects.toThrow("Failed to fetch Umans models info"); }); + it('maps supports_vision sentinel values like "via-handoff" to text-only input', async () => { + const fetchImpl: FetchImpl = async () => + new Response( + JSON.stringify({ + "umans-glm-5.2": { + display_name: "Umans GLM 5.2", + capabilities: { + context_window: 405_504, + max_completion_tokens: 131_071, + recommended_max_tokens: 131_071, + supports_vision: "via-handoff", + supports_tools: true, + reasoning: { supported: true, can_disable: true, default_level: "medium" }, + }, + }, + "umans-coder": { + display_name: "Umans Coder", + capabilities: { + context_window: 262_144, + max_completion_tokens: 262_144, + recommended_max_tokens: 32_768, + supports_vision: true, + supports_tools: true, + reasoning: { supported: true, can_disable: true, default_level: "medium" }, + }, + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + + const fetchDynamicModels = umansModelManagerOptions({ fetch: fetchImpl }).fetchDynamicModels; + if (!fetchDynamicModels) throw new Error("Umans dynamic discovery is not configured"); + + const models = await fetchDynamicModels(); + const glm = models?.find(item => item.id === "umans-glm-5.2"); + const coder = models?.find(item => item.id === "umans-coder"); + + expect(glm?.input).toEqual(["text"]); + expect(coder?.input).toEqual(["text", "image"]); + }); + + it("bundles Umans GLM via-handoff models as text-only", () => { + const providers = modelsJson as Record>; + for (const id of ["umans-glm-5.1", "umans-glm-5.2"] as const) { + const model = providers.umans?.[id]; + expect(model, `${id} should be bundled`).toBeDefined(); + expect(model.input, `${id} input should be text-only`).toEqual(["text"]); + } + }); + + it("drops stale cached GLM rows that predate the via-handoff static correction", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-umans-stale-cache-")); + const dbPath = path.join(tempDir, "models.db"); + const staleGlm: ModelSpec<"anthropic-messages"> = { + id: "umans-glm-5.2", + name: "Umans GLM 5.2", + api: "anthropic-messages", + provider: "umans", + baseUrl: "https://api.code.umans.ai", + reasoning: true, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 405_504, + maxTokens: 131_071, + }; + const correctedGlm: ModelSpec<"anthropic-messages"> = { ...staleGlm, input: ["text"] }; + + try { + await resolveProviderModels( + { + ...umansModelManagerOptions({ + fetch: async () => + new Response( + JSON.stringify({ + "umans-glm-5.2": { + display_name: "Umans GLM 5.2", + capabilities: { + context_window: 405_504, + recommended_max_tokens: 131_071, + supports_vision: true, + supports_tools: true, + reasoning: { supported: true }, + }, + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ), + }), + staticModels: [staleGlm], + cacheDbPath: dbPath, + }, + "online", + ); + + const offline = await resolveProviderModels( + { + ...umansModelManagerOptions({ fetch: async () => new Response(null, { status: 503 }) }), + staticModels: [correctedGlm], + cacheDbPath: dbPath, + }, + "offline", + ); + + const model = offline.models.find(item => item.id === "umans-glm-5.2"); + expect(model?.input).toEqual(["text"]); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("maps the models.dev Umans provider to the Anthropic endpoint", () => { const models = mapModelsDevToModels( {