Merge PR #8200: fix(catalog): Fix reasoning levels of GLM-5.2 models; add Baseten GLM 5.2 Fast (@jcfrancisco)

This commit is contained in:
can1357
2026-08-11 15:09:34 +02:00
7 changed files with 101 additions and 26 deletions
+1
View File
@@ -5,6 +5,7 @@
### Fixed
- Marked `meta/muse-spark-1.2` and `muse-spark-1.2-contributor` as image-capable (`input: ["text", "image"]`) with the same Responses reasoning, thinking, and cost metadata as `muse-spark-1.1` (contributor uses its discounted 0.1/0.2 pricing), so `omp models` no longer lists them as text-only.
- Fixed GLM-5.2 thinking levels across Baseten, CoreWeave, HuggingFace, and other uppercase-ID resellers, which were getting the generic `xhigh` effort ladder instead of the GLM-5.2-specific tiers. Also added Baseten `zai-org/GLM-5.2-Fast` and Fireworks `glm-5.2-fast` as reasoning models ([#8200](https://github.com/can1357/oh-my-pi/pull/8200) by [@jcfrancisco](https://github.com/jcfrancisco)).
## [17.2.12] - 2026-08-08
+3 -3
View File
@@ -143,7 +143,7 @@ export const parseOpenAIModel = parser((modelId): OpenAIModel | null => {
* `parseKnownModel`.
*/
export const parseGlmModel = parser((modelId): GlmModel | null => {
const match = /glm-(\d{1,2}(?:\.\d+)?)(v)?(?:-(air|turbo|flashx|flash|preview))?\b/.exec(modelId);
const match = /glm-(\d{1,2}(?:\.\d+)?)(v)?(?:-(air|turbo|flashx|flash|preview))?\b/i.exec(modelId);
if (!match) {
return null;
}
@@ -153,8 +153,8 @@ export const parseGlmModel = parser((modelId): GlmModel | null => {
}
return {
family: "glm",
variant: (match[3] as GlmVariant | undefined) ?? "base",
vision: match[2] === "v",
variant: (match[3]?.toLowerCase() as GlmVariant | undefined) ?? "base",
vision: match[2]?.toLowerCase() === "v",
version,
};
});
+51 -15
View File
@@ -13793,11 +13793,8 @@
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
"max"
]
},
"supportsComputerUse": false,
@@ -13809,7 +13806,7 @@
"api": "openai-completions",
"provider": "baseten",
"baseUrl": "https://inference.baseten.co/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
@@ -13822,7 +13819,14 @@
"contextWindow": 1048576,
"maxTokens": 262144,
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUseConfig": false,
"thinking": {
"mode": "effort",
"efforts": [
"high",
"max"
]
}
}
},
"bedrock-mantle": {
@@ -16247,7 +16251,7 @@
"low",
"medium",
"high",
"xhigh"
"max"
]
}
}
@@ -21875,6 +21879,40 @@
}
},
"supportsComputerUse": false
},
"glm-5.2-fast": {
"id": "glm-5.2-fast",
"name": "GLM-5.2 Fast",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 2.1,
"output": 6.6,
"cacheRead": 0.21,
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"max"
],
"effortMap": {
"minimal": "none"
}
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
}
},
"github-copilot": {
@@ -28443,7 +28481,7 @@
"low",
"medium",
"high",
"xhigh"
"max"
]
}
}
@@ -89076,7 +89114,7 @@
"low",
"medium",
"high",
"xhigh"
"max"
]
},
"supportsComputerUse": false,
@@ -90094,7 +90132,7 @@
"low",
"medium",
"high",
"xhigh"
"max"
]
}
}
@@ -100992,10 +101030,8 @@
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
"high",
"max"
]
},
"supportsComputerUse": false,
@@ -109424,4 +109460,4 @@
}
}
}
}
}
@@ -1812,6 +1812,7 @@ const FIREWORKS_FAST_VARIANT_SPECS: ReadonlyArray<{
{ base: "kimi-k2.7-code", name: "Kimi K2.7 Code Fast", cost: { input: 1.9, output: 8, cacheRead: 0.38 } },
{ base: "kimi-k2.6", name: "Kimi K2.6 Fast", cost: { input: 2, output: 8, cacheRead: 0.3 } },
{ base: "glm-5.1", name: "GLM-5.1 Fast", cost: { input: 2.8, output: 8.8, cacheRead: 0.52 } },
{ base: "glm-5.2", name: "GLM-5.2 Fast", cost: { input: 2.1, output: 6.6, cacheRead: 0.21 } },
];
/**
@@ -3549,10 +3550,13 @@ export function basetenModelManagerOptions(
const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
const isBasetenNativeReasoning =
// Baseten's reasoning router accepts only the high/max
// effort tiers for its GLM-5.2 and gpt-oss routes.
const isEffortReasoning =
defaults.id === "openai/gpt-oss-120b" ||
defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
defaults.id === "zai-org/GLM-5.2";
defaults.id === "zai-org/GLM-5.2" ||
defaults.id === "zai-org/GLM-5.2-Fast";
const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
const reasoning =
isBasetenNativeReasoning && (features.includes("reasoning") || features.includes("reasoning_effort"));
const supportsTools = features.includes("tools") ? undefined : false;
@@ -3570,10 +3574,6 @@ export function basetenModelManagerOptions(
const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
const baseModel = mapWithBundledReference(entry, defaults, reference);
// Baseten's reasoning router accepts only the high/max
// effort tiers for its GLM-5.2 and gpt-oss routes.
const isEffortReasoning = defaults.id === "openai/gpt-oss-120b" || defaults.id === "zai-org/GLM-5.2";
const thinking = isEffortReasoning
? {
mode: "effort" as const,
@@ -42,6 +42,20 @@ describe("Baseten provider discovery", () => {
input_cache_read: "0.000000145",
},
},
{
id: "zai-org/GLM-5.2-Fast",
object: "model",
name: "GLM 5.2 Fast",
context_length: 524288,
max_completion_tokens: 262144,
supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"],
input_modalities: ["text"],
pricing: {
prompt: "0.0000021",
completion: "0.0000066",
input_cache_read: "0.00000021",
},
},
],
}),
{ status: 200, headers: { "content-type": "application/json" } },
@@ -93,5 +107,17 @@ describe("Baseten provider discovery", () => {
cacheWrite: 0,
},
});
const glmFast = models?.find(model => model.id === "zai-org/GLM-5.2-Fast");
expect(glmFast).toBeDefined();
expect(glmFast).toMatchObject({
provider: "baseten",
api: "openai-completions",
reasoning: true,
thinking: {
mode: "effort",
efforts: ["high", "max"],
},
});
});
});
+2 -1
View File
@@ -24,7 +24,7 @@ describe("buildFireworksFastSeed", () => {
const byId = new Map(seed.map(model => [model.id, model]));
it("emits one fireworks fast variant per curated base", () => {
expect([...byId.keys()].sort()).toEqual(["glm-5.1-fast", "kimi-k2.6-fast", "kimi-k2.7-code-fast"]);
expect([...byId.keys()].sort()).toEqual(["glm-5.1-fast", "glm-5.2-fast", "kimi-k2.6-fast", "kimi-k2.7-code-fast"]);
for (const model of seed) {
expect(model.provider).toBe("fireworks");
expect(isFireworksFastModelId(model.id)).toBe(true);
@@ -35,6 +35,7 @@ describe("buildFireworksFastSeed", () => {
expect(byId.get("kimi-k2.6-fast")?.cost).toEqual({ input: 2, output: 8, cacheRead: 0.3, cacheWrite: 0 });
expect(byId.get("kimi-k2.7-code-fast")?.cost).toEqual({ input: 1.9, output: 8, cacheRead: 0.38, cacheWrite: 0 });
expect(byId.get("glm-5.1-fast")?.cost).toEqual({ input: 2.8, output: 8.8, cacheRead: 0.52, cacheWrite: 0 });
expect(byId.get("glm-5.2-fast")?.cost).toEqual({ input: 2.1, output: 6.6, cacheRead: 0.21, cacheWrite: 0 });
});
it("inherits limits and modalities from the base model", () => {
@@ -227,6 +227,17 @@ describe("isReasoningGlmModelId", () => {
expect(isReasoningGlmModelId("glm-4.5v")).toBe(false);
expect(isReasoningGlmModelId("qwen3.5")).toBe(false);
});
test("matches uppercase provider-prefixed GLM ids", () => {
// Baseten, CoreWeave, HuggingFace, etc. serve GLM under uppercase ids.
expect(isReasoningGlmModelId("zai-org/GLM-5.2")).toBe(true);
expect(isReasoningGlmModelId("zai-org/GLM-5.2-Fast")).toBe(true);
expect(isReasoningGlmModelId("zai-org/GLM-4.7")).toBe(true);
expect(isReasoningGlmModelId("zai-org/GLM-4.5-Air")).toBe(true);
expect(isReasoningGlmModelId("zai-org/GLM-5-Turbo")).toBe(true);
// Vision SKUs are still excluded even in uppercase.
expect(isReasoningGlmModelId("zai-org/GLM-4.5V")).toBe(false);
});
});
describe("isGlmVisionModelId", () => {