feat(catalog): add GLM-5.3 support with uniform low/high/max effort ladder and mandatory thinking
GLM-5.3 introduces three key API changes from GLM-5.2: - Uniform wire-exact low/high/max reasoning_effort ladder on every host (replacing GLM-5.2's host-specific dialects) - Thinking can no longer be disabled (thinking.type must always be "enabled") - Default effort is max Changes: - Add isGlm53ReasoningEffortModelId classifier (>=5.3, base/air/turbo, non-vision) - getModelDefinedEfforts: GLM-5.3 returns LOW_HIGH_MAX uniformly - impliesMandatoryReasoning: GLM-5.3 floors thinking-off to lowest effort - deriveThinking/fillThinkingWireDefaults: defaultLevel=max for GLM-5.3 - generated-policies: pin glm-5.3 to 1M context (zai + zhipu-coding-plan) - descriptors: zai defaultModel -> glm-5.3 - generate-models: curated seed (glm-5.3 is live but not in /models discovery) - models.json: bundled glm-5.3 entry - Tests: catalog thinking-metadata + AI wire-mapping (5 new tests)
This commit is contained in:
@@ -0,0 +1,101 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
// GLM-5.3 replaces GLM-5.2's host-specific reasoning_effort dialects with a
|
||||
// single uniform wire-exact low/high/max ladder on every host, and thinking can
|
||||
// no longer be disabled (thinking.type must always be "enabled"). These tests
|
||||
// pin both contracts so a future change cannot regress to the GLM-5.2 shape.
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "hello", timestamp: Date.now() }],
|
||||
};
|
||||
|
||||
function glm53OnFireworks(): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id: "glm-5.3",
|
||||
name: "GLM-5.3",
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
baseUrl: "https://api.fireworks.ai/inference/v1",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
function glm53OnZaiAnthropic(): Model<"anthropic-messages"> {
|
||||
return buildModel({
|
||||
id: "glm-5.3",
|
||||
name: "GLM-5.3",
|
||||
api: "anthropic-messages",
|
||||
provider: "zai",
|
||||
baseUrl: "https://api.z.ai/api/anthropic",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
} satisfies ModelSpec<"anthropic-messages">);
|
||||
}
|
||||
|
||||
async function captureChatBody(
|
||||
model: Model<"openai-completions">,
|
||||
options: { reasoning?: Effort; disableReasoning?: boolean },
|
||||
): Promise<{ reasoning_effort?: string; thinking?: { type?: string } }> {
|
||||
let requestBody: string | undefined;
|
||||
const fetchMock: FetchImpl = (_input, init) => {
|
||||
requestBody = typeof init?.body === "string" ? init.body : undefined;
|
||||
return Promise.resolve(
|
||||
new Response(
|
||||
'data: {"choices":[{"delta":{"content":"ok"}}]}\ndata: {"choices":[{"finish_reason":"stop"}]}\ndata: [DONE]\n',
|
||||
{ status: 200, headers: { "content-type": "text/event-stream" } },
|
||||
),
|
||||
);
|
||||
};
|
||||
const stream = streamSimple(model, context, { apiKey: "k", fetch: fetchMock, ...options });
|
||||
await stream.result();
|
||||
if (!requestBody) throw new Error("request body was not captured");
|
||||
return JSON.parse(requestBody);
|
||||
}
|
||||
|
||||
describe("GLM-5.3 reasoning effort wire mapping", () => {
|
||||
it("derives the uniform low/high/max ladder on a direct GLM host (not the GLM-5.2 host-specific shape)", () => {
|
||||
const model = glm53OnFireworks();
|
||||
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
expect(model.thinking?.requiresEffort).toBe(true);
|
||||
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
|
||||
});
|
||||
|
||||
it("sends wire-exact low/high/max reasoning_effort on a direct GLM host", async () => {
|
||||
const model = glm53OnFireworks();
|
||||
expect((await captureChatBody(model, { reasoning: Effort.Low })).reasoning_effort).toBe("low");
|
||||
expect((await captureChatBody(model, { reasoning: Effort.High })).reasoning_effort).toBe("high");
|
||||
expect((await captureChatBody(model, { reasoning: Effort.Max })).reasoning_effort).toBe("max");
|
||||
});
|
||||
|
||||
it("clamps thinking-off to the lowest effort instead of disabling (GLM-5.3 cannot disable thinking)", async () => {
|
||||
const model = glm53OnFireworks();
|
||||
const body = await captureChatBody(model, { disableReasoning: true });
|
||||
expect(body.reasoning_effort).toBe("low");
|
||||
expect(body.thinking).toBeUndefined();
|
||||
});
|
||||
|
||||
it("clamps omitted reasoning to the lowest effort", async () => {
|
||||
const model = glm53OnFireworks();
|
||||
const body = await captureChatBody(model, {});
|
||||
expect(body.reasoning_effort).toBe("low");
|
||||
});
|
||||
|
||||
it("derives mandatory reasoning on the zai Anthropic endpoint too", () => {
|
||||
const model = glm53OnZaiAnthropic();
|
||||
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
expect(model.thinking?.requiresEffort).toBe(true);
|
||||
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
|
||||
expect(model.thinking?.mode).toBe("anthropic-budget-effort");
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for GLM-5.3 on the z.AI provider. GLM-5.3 introduces a uniform wire-exact `low`/`high`/`max` reasoning-effort ladder on every host (replacing GLM-5.2's host-specific dialects), makes thinking mandatory (`thinking.type` must always be `enabled`; disabling is no longer supported), and defaults to `max` effort. The model is pinned to 1M context and set as the z.AI provider default.
|
||||
|
||||
## [17.3.2] - 2026-08-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -555,6 +555,24 @@ async function generateModels() {
|
||||
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
|
||||
// applyAnthropicCatalogPolicy.
|
||||
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
|
||||
// Seed GLM-5.3 on the z.AI provider. GLM-5.3 is live on the Anthropic and
|
||||
// coding endpoints but not yet advertised in `/v1/models` (which still tops
|
||||
// out at glm-5.2), so endpoint discovery misses it. The zai provider is not
|
||||
// authoritative, so the seed survives regeneration; thinking metadata
|
||||
// (low/high/max uniform ladder, mandatory reasoning, defaultLevel=max) is
|
||||
// derived by rebakeModelThinking from the identity classifiers.
|
||||
allModels.push({
|
||||
id: "glm-5.3",
|
||||
name: "GLM-5.3",
|
||||
api: "anthropic-messages",
|
||||
provider: "zai",
|
||||
baseUrl: "https://api.z.ai/api/anthropic",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
} as ModelSpec<"anthropic-messages">);
|
||||
// Seed Meta's documented Muse model so first-run selection does not depend on
|
||||
// credentials or live discovery.
|
||||
allModels.push(...META_MUSE_STATIC_MODELS);
|
||||
|
||||
@@ -367,9 +367,13 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
model.omitMaxOutputTokens = true;
|
||||
}
|
||||
|
||||
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
|
||||
// GLM Coding Plan: the selectable 1M-context served ids; pin them so
|
||||
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
|
||||
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
|
||||
// GLM-5.3 succeeds GLM-5.2 with the same 1M context window.
|
||||
if (
|
||||
(model.provider === "zai" || model.provider === "zhipu-coding-plan") &&
|
||||
(model.id === "glm-5.2" || model.id === "glm-5.3")
|
||||
) {
|
||||
model.contextWindow = 1_000_000;
|
||||
model.maxTokens = 131_072;
|
||||
}
|
||||
|
||||
@@ -251,6 +251,25 @@ export const isGlm52ReasoningEffortModelId = memo((modelId: string): boolean =>
|
||||
return semverGte(glm.version, "5.2");
|
||||
});
|
||||
|
||||
/**
|
||||
* GLM-5.3+ coding SKUs. Unlike GLM-5.2 (whose reasoning_effort dialect is
|
||||
* host-specific), GLM-5.3+ exposes a uniform wire-exact `low`/`high`/`max`
|
||||
* ladder on every host, and thinking can no longer be disabled —
|
||||
* `thinking.type` must always be `enabled`. Matching the family keeps future
|
||||
* bumps (`glm-5.4`, `glm-6`, …) covered while excluding the vision (`…v`)
|
||||
* shape and the non-reasoning `-flash`/`-flashx`/`-preview` variants.
|
||||
*/
|
||||
export const isGlm53ReasoningEffortModelId = memo((modelId: string): boolean => {
|
||||
const glm = parseGlmModel(bareModelId(modelId));
|
||||
if (!glm || glm.vision) {
|
||||
return false;
|
||||
}
|
||||
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
|
||||
return false;
|
||||
}
|
||||
return semverGte(glm.version, "5.3");
|
||||
});
|
||||
|
||||
/** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */
|
||||
export const isGlmVisionModelId = memo((modelId: string): boolean => {
|
||||
return parseGlmModel(bareModelId(modelId))?.vision === true;
|
||||
|
||||
@@ -26,6 +26,7 @@ import {
|
||||
isDeepseekModelIdOrName,
|
||||
isDeepseekV4FlashModelId,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGlm53ReasoningEffortModelId,
|
||||
isKimiK3ModelId,
|
||||
isMimoModelIdOrName,
|
||||
isMinimaxM2FamilyModelId,
|
||||
@@ -178,7 +179,8 @@ function fillThinkingWireDefaults<TApi extends Api>(
|
||||
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
||||
supportsAdaptiveThinkingDisplay(spec.id);
|
||||
const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
|
||||
const needsDefaultLevel = thinking.defaultLevel === undefined && isKimiK3ModelId(spec.id);
|
||||
const needsDefaultLevel =
|
||||
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
|
||||
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
|
||||
return thinking;
|
||||
}
|
||||
@@ -216,7 +218,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
|
||||
mode: inferThinkingControlMode(spec, parsed),
|
||||
efforts,
|
||||
};
|
||||
if (isKimiK3ModelId(spec.id)) {
|
||||
if (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)) {
|
||||
config.defaultLevel = Effort.Max;
|
||||
}
|
||||
const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts);
|
||||
@@ -312,6 +314,13 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
spec: ModelSpec<TApi>,
|
||||
compat: CompatOf<TApi>,
|
||||
): readonly Effort[] | undefined {
|
||||
if (isGlm53ReasoningEffortModelId(spec.id)) {
|
||||
// GLM-5.3+ exposes a uniform wire-exact low/high/max ladder on every
|
||||
// host — unlike GLM-5.2, whose reasoning_effort dialect is
|
||||
// host-specific. Thinking can no longer be disabled (handled by
|
||||
// impliesMandatoryReasoning), and the default effort is `max`.
|
||||
return LOW_HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
if (isGlm52ReasoningEffortModelId(spec.id)) {
|
||||
// GLM-5.2's reasoning_effort dialect is host-specific (verified against
|
||||
// live endpoints):
|
||||
@@ -572,6 +581,9 @@ function impliesMandatoryReasoning(parsed: ParsedModel, modelId: string): boolea
|
||||
if (parsed.kind === "pro" && semverGte(parsed.version, "2.5")) return true;
|
||||
}
|
||||
if (isKimiK3ModelId(modelId)) return true;
|
||||
// GLM-5.3+ no longer supports disabling thinking — thinking.type must
|
||||
// always be "enabled". Floor thinking-off requests to the lowest effort.
|
||||
if (isGlm53ReasoningEffortModelId(modelId)) return true;
|
||||
if (isMinimaxM2FamilyModelId(modelId)) return true;
|
||||
if (OPENAI_O_SERIES_RE.test(bareModelId(modelId))) return true;
|
||||
return findThinkingVariantToken(modelId) !== undefined;
|
||||
|
||||
@@ -105719,6 +105719,35 @@
|
||||
]
|
||||
}
|
||||
},
|
||||
"glm-5.3": {
|
||||
"id": "glm-5.3",
|
||||
"name": "GLM-5.3",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "zai",
|
||||
"baseUrl": "https://api.z.ai/api/anthropic",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.4,
|
||||
"output": 4.4,
|
||||
"cacheRead": 0.26,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "anthropic-budget-effort",
|
||||
"efforts": [
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"defaultLevel": "max",
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"glm-5v-turbo": {
|
||||
"id": "glm-5v-turbo",
|
||||
"name": "GLM-5V-Turbo",
|
||||
|
||||
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
|
||||
},
|
||||
{
|
||||
id: "zai",
|
||||
defaultModel: "glm-5.2",
|
||||
defaultModel: "glm-5.3",
|
||||
envVars: ["ZAI_API_KEY"],
|
||||
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
|
||||
catalogDiscovery: { label: "zAI" },
|
||||
|
||||
@@ -240,6 +240,40 @@ describe("generated model policies", () => {
|
||||
expect(models[0]?.maxTokens).toBe(131_072);
|
||||
});
|
||||
|
||||
it("pins zai glm-5.3 to 1M context and derives uniform low/high/max thinking with mandatory reasoning", () => {
|
||||
const models = [
|
||||
createSpec({
|
||||
id: "glm-5.3",
|
||||
api: "anthropic-messages",
|
||||
provider: "zai",
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 8192,
|
||||
}),
|
||||
createSpec({
|
||||
id: "glm-5.3",
|
||||
api: "openai-completions",
|
||||
provider: "zhipu-coding-plan",
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 8192,
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
// Context pinning — same 1M tier as glm-5.2 on both GLM coding-plan hosts.
|
||||
for (const model of models) {
|
||||
expect(model.contextWindow).toBe(1_000_000);
|
||||
expect(model.maxTokens).toBe(131_072);
|
||||
// Uniform wire-exact low/high/max ladder (NOT the host-specific
|
||||
// high/max scale GLM-5.2 uses on zai/zhipu).
|
||||
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
// Thinking can no longer be disabled.
|
||||
expect(model.thinking?.requiresEffort).toBe(true);
|
||||
// Default effort is `max` per the GLM-5.3 API spec.
|
||||
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
|
||||
}
|
||||
});
|
||||
|
||||
it("pins MiniMax-M3 long-context providers to 1M context", () => {
|
||||
const models = [
|
||||
createSpec({
|
||||
|
||||
@@ -63,7 +63,7 @@ function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> {
|
||||
describe("zhipu-coding-plan descriptor", () => {
|
||||
it("defaults to the same Zhipu-hosted model used by login validation", () => {
|
||||
expect(DEFAULT_MODEL_PER_PROVIDER["zhipu-coding-plan"]).toBe("glm-5.1");
|
||||
expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.2");
|
||||
expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.3");
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
Reference in New Issue
Block a user