fix(catalog/umans): treated supports_vision sentinels as text-only
`umans-glm-5.1` / `umans-glm-5.2` advertise themselves on the Umans `models/info` endpoint with `supports_vision: "via-handoff"`. That sentinel means image inputs are routed through a separate vision handoff pre-analysis step; the GLM endpoint itself rejects raw image blocks with `400 This model does not support image inputs`. `umansSupportsVision` was returning `true` for any non-empty string, so dynamic discovery mapped the GLM models to `input: ["text", "image"]` and the agent sent images straight to GLM. The bundled `umans-glm-5.1` / `umans-glm-5.2` rows in `models.json` carried the same stale `["text","image"]` from a previous regen. - Tighten `umansSupportsVision` to `value === true`; document the sentinel contract. - Correct the two bundled rows to `input: ["text"]` so the vision handoff path runs. - Add resolver- and bundle-level regression tests that cover `supports_vision: "via-handoff"` alongside the native-vision `umans-coder` case. Fixes #3184
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Umans `umans-glm-5.1` / `umans-glm-5.2` advertising native image input. The `models/info` endpoint reports `supports_vision: "via-handoff"` for the GLM models, meaning vision routes through a separate handoff pre-analysis step instead of accepting raw image blocks; `umansSupportsVision` treated any non-empty string as native vision support, so image prompts went directly to GLM and were rejected with `400 This model does not support image inputs`. The helper now requires `supports_vision === true`, and the bundled GLM 5.1/5.2 rows are corrected to text-only so the vision-handoff path runs. ([#3184](https://github.com/can1357/oh-my-pi/issues/3184))
|
||||
|
||||
## [16.1.9] - 2026-06-21
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -69284,8 +69284,7 @@
|
||||
"baseUrl": "https://api.code.umans.ai",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
@@ -69327,8 +69326,7 @@
|
||||
]
|
||||
},
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
|
||||
@@ -586,8 +586,18 @@ function normalizeUmansBaseUrl(baseUrl: string | undefined): string {
|
||||
return normalized.endsWith("/v1") ? normalized.slice(0, -3) : normalized;
|
||||
}
|
||||
|
||||
/**
|
||||
* Umans `models/info` reports `supports_vision: true` for natively
|
||||
* vision-capable models and a non-empty string sentinel (e.g.
|
||||
* `"via-handoff"`) for models that route image inputs through a vision
|
||||
* handoff pre-analysis step instead of accepting raw image blocks. Only
|
||||
* `true` means the model accepts image content directly; sentinel values
|
||||
* MUST map to text-only so the agent's vision-handoff path runs instead
|
||||
* of triggering an upstream HTTP 400 (`This model does not support image
|
||||
* inputs`).
|
||||
*/
|
||||
function umansSupportsVision(value: unknown): boolean {
|
||||
return value === true || (typeof value === "string" && value.length > 0);
|
||||
return value === true;
|
||||
}
|
||||
|
||||
function umansReasoningSupported(value: unknown): boolean {
|
||||
|
||||
@@ -101,6 +101,56 @@ describe("umans provider catalog", () => {
|
||||
await expect(fetchDynamicModels()).rejects.toThrow("Failed to fetch Umans models info");
|
||||
});
|
||||
|
||||
it('maps supports_vision sentinel values like "via-handoff" to text-only input', async () => {
|
||||
const fetchImpl: FetchImpl = async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
"umans-glm-5.2": {
|
||||
display_name: "Umans GLM 5.2",
|
||||
capabilities: {
|
||||
context_window: 405_504,
|
||||
max_completion_tokens: 131_071,
|
||||
recommended_max_tokens: 131_071,
|
||||
supports_vision: "via-handoff",
|
||||
supports_tools: true,
|
||||
reasoning: { supported: true, can_disable: true, default_level: "medium" },
|
||||
},
|
||||
},
|
||||
"umans-coder": {
|
||||
display_name: "Umans Coder",
|
||||
capabilities: {
|
||||
context_window: 262_144,
|
||||
max_completion_tokens: 262_144,
|
||||
recommended_max_tokens: 32_768,
|
||||
supports_vision: true,
|
||||
supports_tools: true,
|
||||
reasoning: { supported: true, can_disable: true, default_level: "medium" },
|
||||
},
|
||||
},
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
|
||||
const fetchDynamicModels = umansModelManagerOptions({ fetch: fetchImpl }).fetchDynamicModels;
|
||||
if (!fetchDynamicModels) throw new Error("Umans dynamic discovery is not configured");
|
||||
|
||||
const models = await fetchDynamicModels();
|
||||
const glm = models?.find(item => item.id === "umans-glm-5.2");
|
||||
const coder = models?.find(item => item.id === "umans-coder");
|
||||
|
||||
expect(glm?.input).toEqual(["text"]);
|
||||
expect(coder?.input).toEqual(["text", "image"]);
|
||||
});
|
||||
|
||||
it("bundles Umans GLM via-handoff models as text-only", () => {
|
||||
const providers = modelsJson as Record<string, Record<string, BundledModel>>;
|
||||
for (const id of ["umans-glm-5.1", "umans-glm-5.2"] as const) {
|
||||
const model = providers.umans?.[id];
|
||||
expect(model, `${id} should be bundled`).toBeDefined();
|
||||
expect(model.input, `${id} input should be text-only`).toEqual(["text"]);
|
||||
}
|
||||
});
|
||||
|
||||
it("maps the models.dev Umans provider to the Anthropic endpoint", () => {
|
||||
const models = mapModelsDevToModels(
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user