feat(catalog): add GLM-5.3 support with uniform low/high/max effort ladder and mandatory thinking

GLM-5.3 introduces three key API changes from GLM-5.2:
- Uniform wire-exact low/high/max reasoning_effort ladder on every host
  (replacing GLM-5.2's host-specific dialects)
- Thinking can no longer be disabled (thinking.type must always be "enabled")
- Default effort is max

Changes:
- Add isGlm53ReasoningEffortModelId classifier (>=5.3, base/air/turbo, non-vision)
- getModelDefinedEfforts: GLM-5.3 returns LOW_HIGH_MAX uniformly
- impliesMandatoryReasoning: GLM-5.3 floors thinking-off to lowest effort
- deriveThinking/fillThinkingWireDefaults: defaultLevel=max for GLM-5.3
- generated-policies: pin glm-5.3 to 1M context (zai + zhipu-coding-plan)
- descriptors: zai defaultModel -> glm-5.3
- generate-models: curated seed (glm-5.3 is live but not in /models discovery)
- models.json: bundled glm-5.3 entry
- Tests: catalog thinking-metadata + AI wire-mapping (5 new tests)
This commit is contained in:
oldschoola
2026-08-13 23:18:17 -07:00
parent ad318c7572
commit e49ee4b4e2
10 changed files with 227 additions and 6 deletions
@@ -0,0 +1,101 @@
import { describe, expect, it } from "bun:test";
import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
// GLM-5.3 replaces GLM-5.2's host-specific reasoning_effort dialects with a
// single uniform wire-exact low/high/max ladder on every host, and thinking can
// no longer be disabled (thinking.type must always be "enabled"). These tests
// pin both contracts so a future change cannot regress to the GLM-5.2 shape.
const context: Context = {
messages: [{ role: "user", content: "hello", timestamp: Date.now() }],
};
function glm53OnFireworks(): Model<"openai-completions"> {
return buildModel({
id: "glm-5.3",
name: "GLM-5.3",
api: "openai-completions",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"openai-completions">);
}
function glm53OnZaiAnthropic(): Model<"anthropic-messages"> {
return buildModel({
id: "glm-5.3",
name: "GLM-5.3",
api: "anthropic-messages",
provider: "zai",
baseUrl: "https://api.z.ai/api/anthropic",
reasoning: true,
input: ["text"],
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"anthropic-messages">);
}
async function captureChatBody(
model: Model<"openai-completions">,
options: { reasoning?: Effort; disableReasoning?: boolean },
): Promise<{ reasoning_effort?: string; thinking?: { type?: string } }> {
let requestBody: string | undefined;
const fetchMock: FetchImpl = (_input, init) => {
requestBody = typeof init?.body === "string" ? init.body : undefined;
return Promise.resolve(
new Response(
'data: {"choices":[{"delta":{"content":"ok"}}]}\ndata: {"choices":[{"finish_reason":"stop"}]}\ndata: [DONE]\n',
{ status: 200, headers: { "content-type": "text/event-stream" } },
),
);
};
const stream = streamSimple(model, context, { apiKey: "k", fetch: fetchMock, ...options });
await stream.result();
if (!requestBody) throw new Error("request body was not captured");
return JSON.parse(requestBody);
}
describe("GLM-5.3 reasoning effort wire mapping", () => {
it("derives the uniform low/high/max ladder on a direct GLM host (not the GLM-5.2 host-specific shape)", () => {
const model = glm53OnFireworks();
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
expect(model.thinking?.requiresEffort).toBe(true);
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
});
it("sends wire-exact low/high/max reasoning_effort on a direct GLM host", async () => {
const model = glm53OnFireworks();
expect((await captureChatBody(model, { reasoning: Effort.Low })).reasoning_effort).toBe("low");
expect((await captureChatBody(model, { reasoning: Effort.High })).reasoning_effort).toBe("high");
expect((await captureChatBody(model, { reasoning: Effort.Max })).reasoning_effort).toBe("max");
});
it("clamps thinking-off to the lowest effort instead of disabling (GLM-5.3 cannot disable thinking)", async () => {
const model = glm53OnFireworks();
const body = await captureChatBody(model, { disableReasoning: true });
expect(body.reasoning_effort).toBe("low");
expect(body.thinking).toBeUndefined();
});
it("clamps omitted reasoning to the lowest effort", async () => {
const model = glm53OnFireworks();
const body = await captureChatBody(model, {});
expect(body.reasoning_effort).toBe("low");
});
it("derives mandatory reasoning on the zai Anthropic endpoint too", () => {
const model = glm53OnZaiAnthropic();
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
expect(model.thinking?.requiresEffort).toBe(true);
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
expect(model.thinking?.mode).toBe("anthropic-budget-effort");
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added support for GLM-5.3 on the z.AI provider. GLM-5.3 introduces a uniform wire-exact `low`/`high`/`max` reasoning-effort ladder on every host (replacing GLM-5.2's host-specific dialects), makes thinking mandatory (`thinking.type` must always be `enabled`; disabling is no longer supported), and defaults to `max` effort. The model is pinned to 1M context and set as the z.AI provider default.
## [17.3.2] - 2026-08-13
### Added
@@ -555,6 +555,24 @@ async function generateModels() {
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
// applyAnthropicCatalogPolicy.
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
// Seed GLM-5.3 on the z.AI provider. GLM-5.3 is live on the Anthropic and
// coding endpoints but not yet advertised in `/v1/models` (which still tops
// out at glm-5.2), so endpoint discovery misses it. The zai provider is not
// authoritative, so the seed survives regeneration; thinking metadata
// (low/high/max uniform ladder, mandatory reasoning, defaultLevel=max) is
// derived by rebakeModelThinking from the identity classifiers.
allModels.push({
id: "glm-5.3",
name: "GLM-5.3",
api: "anthropic-messages",
provider: "zai",
baseUrl: "https://api.z.ai/api/anthropic",
reasoning: true,
input: ["text"],
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} as ModelSpec<"anthropic-messages">);
// Seed Meta's documented Muse model so first-run selection does not depend on
// credentials or live discovery.
allModels.push(...META_MUSE_STATIC_MODELS);
@@ -367,9 +367,13 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.omitMaxOutputTokens = true;
}
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
// GLM Coding Plan: the selectable 1M-context served ids; pin them so
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
// GLM-5.3 succeeds GLM-5.2 with the same 1M context window.
if (
(model.provider === "zai" || model.provider === "zhipu-coding-plan") &&
(model.id === "glm-5.2" || model.id === "glm-5.3")
) {
model.contextWindow = 1_000_000;
model.maxTokens = 131_072;
}
+19
View File
@@ -251,6 +251,25 @@ export const isGlm52ReasoningEffortModelId = memo((modelId: string): boolean =>
return semverGte(glm.version, "5.2");
});
/**
* GLM-5.3+ coding SKUs. Unlike GLM-5.2 (whose reasoning_effort dialect is
* host-specific), GLM-5.3+ exposes a uniform wire-exact `low`/`high`/`max`
* ladder on every host, and thinking can no longer be disabled —
* `thinking.type` must always be `enabled`. Matching the family keeps future
* bumps (`glm-5.4`, `glm-6`, …) covered while excluding the vision (`…v`)
* shape and the non-reasoning `-flash`/`-flashx`/`-preview` variants.
*/
export const isGlm53ReasoningEffortModelId = memo((modelId: string): boolean => {
const glm = parseGlmModel(bareModelId(modelId));
if (!glm || glm.vision) {
return false;
}
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
return false;
}
return semverGte(glm.version, "5.3");
});
/** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */
export const isGlmVisionModelId = memo((modelId: string): boolean => {
return parseGlmModel(bareModelId(modelId))?.vision === true;
+14 -2
View File
@@ -26,6 +26,7 @@ import {
isDeepseekModelIdOrName,
isDeepseekV4FlashModelId,
isGlm52ReasoningEffortModelId,
isGlm53ReasoningEffortModelId,
isKimiK3ModelId,
isMimoModelIdOrName,
isMinimaxM2FamilyModelId,
@@ -178,7 +179,8 @@ function fillThinkingWireDefaults<TApi extends Api>(
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
supportsAdaptiveThinkingDisplay(spec.id);
const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
const needsDefaultLevel = thinking.defaultLevel === undefined && isKimiK3ModelId(spec.id);
const needsDefaultLevel =
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
return thinking;
}
@@ -216,7 +218,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
mode: inferThinkingControlMode(spec, parsed),
efforts,
};
if (isKimiK3ModelId(spec.id)) {
if (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)) {
config.defaultLevel = Effort.Max;
}
const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts);
@@ -312,6 +314,13 @@ function getModelDefinedEfforts<TApi extends Api>(
spec: ModelSpec<TApi>,
compat: CompatOf<TApi>,
): readonly Effort[] | undefined {
if (isGlm53ReasoningEffortModelId(spec.id)) {
// GLM-5.3+ exposes a uniform wire-exact low/high/max ladder on every
// host — unlike GLM-5.2, whose reasoning_effort dialect is
// host-specific. Thinking can no longer be disabled (handled by
// impliesMandatoryReasoning), and the default effort is `max`.
return LOW_HIGH_MAX_REASONING_EFFORTS;
}
if (isGlm52ReasoningEffortModelId(spec.id)) {
// GLM-5.2's reasoning_effort dialect is host-specific (verified against
// live endpoints):
@@ -572,6 +581,9 @@ function impliesMandatoryReasoning(parsed: ParsedModel, modelId: string): boolea
if (parsed.kind === "pro" && semverGte(parsed.version, "2.5")) return true;
}
if (isKimiK3ModelId(modelId)) return true;
// GLM-5.3+ no longer supports disabling thinking — thinking.type must
// always be "enabled". Floor thinking-off requests to the lowest effort.
if (isGlm53ReasoningEffortModelId(modelId)) return true;
if (isMinimaxM2FamilyModelId(modelId)) return true;
if (OPENAI_O_SERIES_RE.test(bareModelId(modelId))) return true;
return findThinkingVariantToken(modelId) !== undefined;
+29
View File
@@ -105719,6 +105719,35 @@
]
}
},
"glm-5.3": {
"id": "glm-5.3",
"name": "GLM-5.3",
"api": "anthropic-messages",
"provider": "zai",
"baseUrl": "https://api.z.ai/api/anthropic",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.4,
"output": 4.4,
"cacheRead": 0.26,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 131072,
"thinking": {
"mode": "anthropic-budget-effort",
"efforts": [
"low",
"high",
"max"
],
"defaultLevel": "max",
"requiresEffort": true
}
},
"glm-5v-turbo": {
"id": "glm-5v-turbo",
"name": "GLM-5V-Turbo",
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "zai",
defaultModel: "glm-5.2",
defaultModel: "glm-5.3",
envVars: ["ZAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
catalogDiscovery: { label: "zAI" },
@@ -240,6 +240,40 @@ describe("generated model policies", () => {
expect(models[0]?.maxTokens).toBe(131_072);
});
it("pins zai glm-5.3 to 1M context and derives uniform low/high/max thinking with mandatory reasoning", () => {
const models = [
createSpec({
id: "glm-5.3",
api: "anthropic-messages",
provider: "zai",
contextWindow: 200_000,
maxTokens: 8192,
}),
createSpec({
id: "glm-5.3",
api: "openai-completions",
provider: "zhipu-coding-plan",
contextWindow: 200_000,
maxTokens: 8192,
}),
];
applyGeneratedModelPolicies(models);
// Context pinning — same 1M tier as glm-5.2 on both GLM coding-plan hosts.
for (const model of models) {
expect(model.contextWindow).toBe(1_000_000);
expect(model.maxTokens).toBe(131_072);
// Uniform wire-exact low/high/max ladder (NOT the host-specific
// high/max scale GLM-5.2 uses on zai/zhipu).
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
// Thinking can no longer be disabled.
expect(model.thinking?.requiresEffort).toBe(true);
// Default effort is `max` per the GLM-5.3 API spec.
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
}
});
it("pins MiniMax-M3 long-context providers to 1M context", () => {
const models = [
createSpec({
+1 -1
View File
@@ -63,7 +63,7 @@ function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> {
describe("zhipu-coding-plan descriptor", () => {
it("defaults to the same Zhipu-hosted model used by login validation", () => {
expect(DEFAULT_MODEL_PER_PROVIDER["zhipu-coding-plan"]).toBe("glm-5.1");
expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.2");
expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.3");
});
});