fix(catalog/umans): treated supports_vision sentinels as text-only

`umans-glm-5.1` / `umans-glm-5.2` advertise themselves on the Umans
`models/info` endpoint with `supports_vision: "via-handoff"`. That
sentinel means image inputs are routed through a separate vision
handoff pre-analysis step; the GLM endpoint itself rejects raw image
blocks with `400 This model does not support image inputs`.

`umansSupportsVision` was returning `true` for any non-empty string,
so dynamic discovery mapped the GLM models to `input: ["text",
"image"]` and the agent sent images straight to GLM. The bundled
`umans-glm-5.1` / `umans-glm-5.2` rows in `models.json` carried the
same stale `["text","image"]` from a previous regen.

- Tighten `umansSupportsVision` to `value === true`; document the
  sentinel contract.
- Correct the two bundled rows to `input: ["text"]` so the vision
  handoff path runs.
- Add resolver- and bundle-level regression tests that cover
  `supports_vision: "via-handoff"` alongside the native-vision
  `umans-coder` case.

Fixes #3184
This commit is contained in:
roboomp
2026-06-21 10:32:53 +00:00
parent 35508c48f7
commit 565aba94bc
4 changed files with 67 additions and 5 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Umans `umans-glm-5.1` / `umans-glm-5.2` advertising native image input. The `models/info` endpoint reports `supports_vision: "via-handoff"` for the GLM models, meaning vision routes through a separate handoff pre-analysis step instead of accepting raw image blocks; `umansSupportsVision` treated any non-empty string as native vision support, so image prompts went directly to GLM and were rejected with `400 This model does not support image inputs`. The helper now requires `supports_vision === true`, and the bundled GLM 5.1/5.2 rows are corrected to text-only so the vision-handoff path runs. ([#3184](https://github.com/can1357/oh-my-pi/issues/3184))
## [16.1.9] - 2026-06-21
### Fixed
+2 -4
View File
@@ -69284,8 +69284,7 @@
"baseUrl": "https://api.code.umans.ai",
"reasoning": true,
"input": [
"text",
"image"
"text"
],
"cost": {
"input": 0,
@@ -69327,8 +69326,7 @@
]
},
"input": [
"text",
"image"
"text"
],
"cost": {
"input": 0,
@@ -586,8 +586,18 @@ function normalizeUmansBaseUrl(baseUrl: string | undefined): string {
return normalized.endsWith("/v1") ? normalized.slice(0, -3) : normalized;
}
/**
* Umans `models/info` reports `supports_vision: true` for natively
* vision-capable models and a non-empty string sentinel (e.g.
* `"via-handoff"`) for models that route image inputs through a vision
* handoff pre-analysis step instead of accepting raw image blocks. Only
* `true` means the model accepts image content directly; sentinel values
* MUST map to text-only so the agent's vision-handoff path runs instead
* of triggering an upstream HTTP 400 (`This model does not support image
* inputs`).
*/
function umansSupportsVision(value: unknown): boolean {
return value === true || (typeof value === "string" && value.length > 0);
return value === true;
}
function umansReasoningSupported(value: unknown): boolean {
@@ -101,6 +101,56 @@ describe("umans provider catalog", () => {
await expect(fetchDynamicModels()).rejects.toThrow("Failed to fetch Umans models info");
});
it('maps supports_vision sentinel values like "via-handoff" to text-only input', async () => {
const fetchImpl: FetchImpl = async () =>
new Response(
JSON.stringify({
"umans-glm-5.2": {
display_name: "Umans GLM 5.2",
capabilities: {
context_window: 405_504,
max_completion_tokens: 131_071,
recommended_max_tokens: 131_071,
supports_vision: "via-handoff",
supports_tools: true,
reasoning: { supported: true, can_disable: true, default_level: "medium" },
},
},
"umans-coder": {
display_name: "Umans Coder",
capabilities: {
context_window: 262_144,
max_completion_tokens: 262_144,
recommended_max_tokens: 32_768,
supports_vision: true,
supports_tools: true,
reasoning: { supported: true, can_disable: true, default_level: "medium" },
},
},
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
const fetchDynamicModels = umansModelManagerOptions({ fetch: fetchImpl }).fetchDynamicModels;
if (!fetchDynamicModels) throw new Error("Umans dynamic discovery is not configured");
const models = await fetchDynamicModels();
const glm = models?.find(item => item.id === "umans-glm-5.2");
const coder = models?.find(item => item.id === "umans-coder");
expect(glm?.input).toEqual(["text"]);
expect(coder?.input).toEqual(["text", "image"]);
});
it("bundles Umans GLM via-handoff models as text-only", () => {
const providers = modelsJson as Record<string, Record<string, BundledModel>>;
for (const id of ["umans-glm-5.1", "umans-glm-5.2"] as const) {
const model = providers.umans?.[id];
expect(model, `${id} should be bundled`).toBeDefined();
expect(model.input, `${id} input should be text-only`).toEqual(["text"]);
}
});
it("maps the models.dev Umans provider to the Anthropic endpoint", () => {
const models = mapModelsDevToModels(
{