Merge PR #3186: fix(catalog/umans): treated supports_vision sentinels as text-only (@roboomp)

This commit is contained in:
can1357
2026-06-21 16:28:18 +02:00
5 changed files with 160 additions and 8 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Umans `umans-glm-5.1` / `umans-glm-5.2` advertising native image input. The `models/info` endpoint reports `supports_vision: "via-handoff"` for the GLM models, meaning vision routes through a separate handoff pre-analysis step instead of accepting raw image blocks; `umansSupportsVision` treated any non-empty string as native vision support, so image prompts went directly to GLM and were rejected with `400 This model does not support image inputs`. The helper now requires `supports_vision === true`, the bundled GLM 5.1/5.2 rows are corrected to text-only, and stale mismatched Umans cache rows for those ids are dropped so the vision-handoff path runs even before a successful refresh. ([#3184](https://github.com/can1357/oh-my-pi/issues/3184))
## [16.1.9] - 2026-06-21
### Fixed
+26 -2
View File
@@ -39,6 +39,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
cacheTtlMs?: number;
/** When true, a successful dynamic fetch is the complete provider catalog and prunes static-only models. */
dynamicModelsAuthoritative?: boolean;
/** Cached model ids to ignore when the cache was written against a different static catalog fingerprint. */
dropCachedModelIdsOnStaticMismatch?: readonly string[];
/** Optional dynamic endpoint fetcher. */
fetchDynamicModels?: () => Promise<readonly ModelSpec<TApi>[] | null>;
/** Optional models.dev fallback hook. */
@@ -148,7 +150,13 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
const shouldUseFreshCacheAsAuthoritative =
strategy === "online-if-uncached" && hasUsableFreshCache && hasAuthoritativeCache;
const dynamicFetchSucceeded = fetchedDynamicModels !== null;
const cacheModels = dynamicFetchSucceeded ? [] : normalizeModelList<TApi>(cache?.models ?? []);
const cacheModels = dynamicFetchSucceeded
? []
: dropCachedModelIdsOnStaticMismatch(
normalizeModelList<TApi>(cache?.models ?? []),
cacheFingerprintMatches,
options.dropCachedModelIdsOnStaticMismatch,
);
const dynamicModels = fetchedDynamicModels ?? [];
const mergedWithCache = mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), cacheModels);
const mergedModels = mergeDynamicModels(mergedWithCache, dynamicModels);
@@ -180,7 +188,11 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
collapseBuiltModelVariants(
mergeDynamicModels(
mergeModelSources(staticModels, modelsDevModels),
normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
dropCachedModelIdsOnStaticMismatch(
normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
cacheFingerprintMatches,
options.dropCachedModelIdsOnStaticMismatch,
),
),
),
false,
@@ -248,6 +260,18 @@ function shouldFetchRemoteSources(
return false;
}
function dropCachedModelIdsOnStaticMismatch<TApi extends Api>(
models: readonly Model<TApi>[],
cacheFingerprintMatches: boolean,
ids: readonly string[] | undefined,
): Model<TApi>[] {
if (cacheFingerprintMatches || ids === undefined || ids.length === 0 || models.length === 0) {
return models.length === 0 ? [] : [...models];
}
const droppedIds = new Set(ids);
return models.filter(model => !droppedIds.has(model.id));
}
function mergeModelSources<TApi extends Api>(...sources: readonly (readonly Model<TApi>[])[]): Model<TApi>[] {
// Strip out empty/missing sources up front. The hot path is `(static, [])`
// (modelsDev disabled / failed) — a single non-empty source means we can
+2 -4
View File
@@ -69284,8 +69284,7 @@
"baseUrl": "https://api.code.umans.ai",
"reasoning": true,
"input": [
"text",
"image"
"text"
],
"cost": {
"input": 0,
@@ -69327,8 +69326,7 @@
]
},
"input": [
"text",
"image"
"text"
],
"cost": {
"input": 0,
@@ -568,6 +568,7 @@ const UMANS_REASONING_EFFORT_BY_LEVEL: Record<string, Effort> = {
xhigh: Effort.XHigh,
};
const UMANS_DEFAULT_REASONING_EFFORTS = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const;
const UMANS_VIA_HANDOFF_MODEL_IDS = ["umans-glm-5.1", "umans-glm-5.2"] as const;
export interface UmansModelManagerConfig {
apiKey?: string;
@@ -586,8 +587,18 @@ function normalizeUmansBaseUrl(baseUrl: string | undefined): string {
return normalized.endsWith("/v1") ? normalized.slice(0, -3) : normalized;
}
/**
* Umans `models/info` reports `supports_vision: true` for natively
* vision-capable models and a non-empty string sentinel (e.g.
* `"via-handoff"`) for models that route image inputs through a vision
* handoff pre-analysis step instead of accepting raw image blocks. Only
* `true` means the model accepts image content directly; sentinel values
* MUST map to text-only so the agent's vision-handoff path runs instead
* of triggering an upstream HTTP 400 (`This model does not support image
* inputs`).
*/
function umansSupportsVision(value: unknown): boolean {
return value === true || (typeof value === "string" && value.length > 0);
return value === true;
}
function umansReasoningSupported(value: unknown): boolean {
@@ -704,6 +715,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
return {
providerId: "umans",
dynamicModelsAuthoritative: true,
dropCachedModelIdsOnStaticMismatch: UMANS_VIA_HANDOFF_MODEL_IDS,
fetchDynamicModels: () => fetchUmansModelsInfo({ baseUrl, apiKey, fetch: config?.fetch, references }),
};
}
+115 -1
View File
@@ -1,10 +1,14 @@
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager";
import {
MODELS_DEV_PROVIDER_DESCRIPTORS,
mapModelsDevToModels,
umansModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types";
import modelsJson from "../src/models.json";
interface BundledModel {
@@ -101,6 +105,116 @@ describe("umans provider catalog", () => {
await expect(fetchDynamicModels()).rejects.toThrow("Failed to fetch Umans models info");
});
it('maps supports_vision sentinel values like "via-handoff" to text-only input', async () => {
const fetchImpl: FetchImpl = async () =>
new Response(
JSON.stringify({
"umans-glm-5.2": {
display_name: "Umans GLM 5.2",
capabilities: {
context_window: 405_504,
max_completion_tokens: 131_071,
recommended_max_tokens: 131_071,
supports_vision: "via-handoff",
supports_tools: true,
reasoning: { supported: true, can_disable: true, default_level: "medium" },
},
},
"umans-coder": {
display_name: "Umans Coder",
capabilities: {
context_window: 262_144,
max_completion_tokens: 262_144,
recommended_max_tokens: 32_768,
supports_vision: true,
supports_tools: true,
reasoning: { supported: true, can_disable: true, default_level: "medium" },
},
},
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
const fetchDynamicModels = umansModelManagerOptions({ fetch: fetchImpl }).fetchDynamicModels;
if (!fetchDynamicModels) throw new Error("Umans dynamic discovery is not configured");
const models = await fetchDynamicModels();
const glm = models?.find(item => item.id === "umans-glm-5.2");
const coder = models?.find(item => item.id === "umans-coder");
expect(glm?.input).toEqual(["text"]);
expect(coder?.input).toEqual(["text", "image"]);
});
it("bundles Umans GLM via-handoff models as text-only", () => {
const providers = modelsJson as Record<string, Record<string, BundledModel>>;
for (const id of ["umans-glm-5.1", "umans-glm-5.2"] as const) {
const model = providers.umans?.[id];
expect(model, `${id} should be bundled`).toBeDefined();
expect(model.input, `${id} input should be text-only`).toEqual(["text"]);
}
});
it("drops stale cached GLM rows that predate the via-handoff static correction", async () => {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-umans-stale-cache-"));
const dbPath = path.join(tempDir, "models.db");
const staleGlm: ModelSpec<"anthropic-messages"> = {
id: "umans-glm-5.2",
name: "Umans GLM 5.2",
api: "anthropic-messages",
provider: "umans",
baseUrl: "https://api.code.umans.ai",
reasoning: true,
input: ["text", "image"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 405_504,
maxTokens: 131_071,
};
const correctedGlm: ModelSpec<"anthropic-messages"> = { ...staleGlm, input: ["text"] };
try {
await resolveProviderModels(
{
...umansModelManagerOptions({
fetch: async () =>
new Response(
JSON.stringify({
"umans-glm-5.2": {
display_name: "Umans GLM 5.2",
capabilities: {
context_window: 405_504,
recommended_max_tokens: 131_071,
supports_vision: true,
supports_tools: true,
reasoning: { supported: true },
},
},
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
),
}),
staticModels: [staleGlm],
cacheDbPath: dbPath,
},
"online",
);
const offline = await resolveProviderModels(
{
...umansModelManagerOptions({ fetch: async () => new Response(null, { status: 503 }) }),
staticModels: [correctedGlm],
cacheDbPath: dbPath,
},
"offline",
);
const model = offline.models.find(item => item.id === "umans-glm-5.2");
expect(model?.input).toEqual(["text"]);
} finally {
await fs.rm(tempDir, { recursive: true, force: true });
}
});
it("maps the models.dev Umans provider to the Anthropic endpoint", () => {
const models = mapModelsDevToModels(
{