diff --git a/packages/ai/test/glm-5.3-reasoning-effort.test.ts b/packages/ai/test/glm-5.3-reasoning-effort.test.ts new file mode 100644 index 000000000..158507292 --- /dev/null +++ b/packages/ai/test/glm-5.3-reasoning-effort.test.ts @@ -0,0 +1,101 @@ +import { describe, expect, it } from "bun:test"; +import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai"; +import { streamSimple } from "@oh-my-pi/pi-ai/stream"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; + +// GLM-5.3 replaces GLM-5.2's host-specific reasoning_effort dialects with a +// single uniform wire-exact low/high/max ladder on every host, and thinking can +// no longer be disabled (thinking.type must always be "enabled"). These tests +// pin both contracts so a future change cannot regress to the GLM-5.2 shape. +const context: Context = { + messages: [{ role: "user", content: "hello", timestamp: Date.now() }], +}; + +function glm53OnFireworks(): Model<"openai-completions"> { + return buildModel({ + id: "glm-5.3", + name: "GLM-5.3", + api: "openai-completions", + provider: "fireworks", + baseUrl: "https://api.fireworks.ai/inference/v1", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 131_072, + } satisfies ModelSpec<"openai-completions">); +} + +function glm53OnZaiAnthropic(): Model<"anthropic-messages"> { + return buildModel({ + id: "glm-5.3", + name: "GLM-5.3", + api: "anthropic-messages", + provider: "zai", + baseUrl: "https://api.z.ai/api/anthropic", + reasoning: true, + input: ["text"], + cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 131_072, + } satisfies ModelSpec<"anthropic-messages">); +} + +async function captureChatBody( + model: Model<"openai-completions">, + options: { reasoning?: Effort; disableReasoning?: boolean }, +): Promise<{ reasoning_effort?: string; thinking?: { type?: string } }> { + let requestBody: string | undefined; + const fetchMock: FetchImpl = (_input, init) => { + requestBody = typeof init?.body === "string" ? init.body : undefined; + return Promise.resolve( + new Response( + 'data: {"choices":[{"delta":{"content":"ok"}}]}\ndata: {"choices":[{"finish_reason":"stop"}]}\ndata: [DONE]\n', + { status: 200, headers: { "content-type": "text/event-stream" } }, + ), + ); + }; + const stream = streamSimple(model, context, { apiKey: "k", fetch: fetchMock, ...options }); + await stream.result(); + if (!requestBody) throw new Error("request body was not captured"); + return JSON.parse(requestBody); +} + +describe("GLM-5.3 reasoning effort wire mapping", () => { + it("derives the uniform low/high/max ladder on a direct GLM host (not the GLM-5.2 host-specific shape)", () => { + const model = glm53OnFireworks(); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]); + expect(model.thinking?.requiresEffort).toBe(true); + expect(model.thinking?.defaultLevel).toBe(Effort.Max); + }); + + it("sends wire-exact low/high/max reasoning_effort on a direct GLM host", async () => { + const model = glm53OnFireworks(); + expect((await captureChatBody(model, { reasoning: Effort.Low })).reasoning_effort).toBe("low"); + expect((await captureChatBody(model, { reasoning: Effort.High })).reasoning_effort).toBe("high"); + expect((await captureChatBody(model, { reasoning: Effort.Max })).reasoning_effort).toBe("max"); + }); + + it("clamps thinking-off to the lowest effort instead of disabling (GLM-5.3 cannot disable thinking)", async () => { + const model = glm53OnFireworks(); + const body = await captureChatBody(model, { disableReasoning: true }); + expect(body.reasoning_effort).toBe("low"); + expect(body.thinking).toBeUndefined(); + }); + + it("clamps omitted reasoning to the lowest effort", async () => { + const model = glm53OnFireworks(); + const body = await captureChatBody(model, {}); + expect(body.reasoning_effort).toBe("low"); + }); + + it("derives mandatory reasoning on the zai Anthropic endpoint too", () => { + const model = glm53OnZaiAnthropic(); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]); + expect(model.thinking?.requiresEffort).toBe(true); + expect(model.thinking?.defaultLevel).toBe(Effort.Max); + expect(model.thinking?.mode).toBe("anthropic-budget-effort"); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 8cde7ddb0..d8b1d11be 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added support for GLM-5.3 on the z.AI provider. GLM-5.3 introduces a uniform wire-exact `low`/`high`/`max` reasoning-effort ladder on every host (replacing GLM-5.2's host-specific dialects), makes thinking mandatory (`thinking.type` must always be `enabled`; disabling is no longer supported), and defaults to `max` effort. The model is pinned to 1M context and set as the z.AI provider default. + ## [17.3.2] - 2026-08-13 ### Added diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index e51cabdf4..bd1d0e914 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -555,6 +555,24 @@ async function generateModels() { // Mythos 5). Deduped behind upstream entries; metadata is pinned in // applyAnthropicCatalogPolicy. allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS); + // Seed GLM-5.3 on the z.AI provider. GLM-5.3 is live on the Anthropic and + // coding endpoints but not yet advertised in `/v1/models` (which still tops + // out at glm-5.2), so endpoint discovery misses it. The zai provider is not + // authoritative, so the seed survives regeneration; thinking metadata + // (low/high/max uniform ladder, mandatory reasoning, defaultLevel=max) is + // derived by rebakeModelThinking from the identity classifiers. + allModels.push({ + id: "glm-5.3", + name: "GLM-5.3", + api: "anthropic-messages", + provider: "zai", + baseUrl: "https://api.z.ai/api/anthropic", + reasoning: true, + input: ["text"], + cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 131_072, + } as ModelSpec<"anthropic-messages">); // Seed Meta's documented Muse model so first-run selection does not depend on // credentials or live discovery. allModels.push(...META_MUSE_STATIC_MODELS); diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 2b97e0dfb..abf8f82cc 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -367,9 +367,13 @@ function applyGeneratedModelPolicy(model: ModelSpec): void { model.omitMaxOutputTokens = true; } - // GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so + // GLM Coding Plan: the selectable 1M-context served ids; pin them so // endpoint discovery or older bundled fallbacks cannot regress to 200k. - if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") { + // GLM-5.3 succeeds GLM-5.2 with the same 1M context window. + if ( + (model.provider === "zai" || model.provider === "zhipu-coding-plan") && + (model.id === "glm-5.2" || model.id === "glm-5.3") + ) { model.contextWindow = 1_000_000; model.maxTokens = 131_072; } diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index 289f80821..e57ca16f7 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -251,6 +251,25 @@ export const isGlm52ReasoningEffortModelId = memo((modelId: string): boolean => return semverGte(glm.version, "5.2"); }); +/** + * GLM-5.3+ coding SKUs. Unlike GLM-5.2 (whose reasoning_effort dialect is + * host-specific), GLM-5.3+ exposes a uniform wire-exact `low`/`high`/`max` + * ladder on every host, and thinking can no longer be disabled — + * `thinking.type` must always be `enabled`. Matching the family keeps future + * bumps (`glm-5.4`, `glm-6`, …) covered while excluding the vision (`…v`) + * shape and the non-reasoning `-flash`/`-flashx`/`-preview` variants. + */ +export const isGlm53ReasoningEffortModelId = memo((modelId: string): boolean => { + const glm = parseGlmModel(bareModelId(modelId)); + if (!glm || glm.vision) { + return false; + } + if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") { + return false; + } + return semverGte(glm.version, "5.3"); +}); + /** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */ export const isGlmVisionModelId = memo((modelId: string): boolean => { return parseGlmModel(bareModelId(modelId))?.vision === true; diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index f83237574..7298d6217 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -26,6 +26,7 @@ import { isDeepseekModelIdOrName, isDeepseekV4FlashModelId, isGlm52ReasoningEffortModelId, + isGlm53ReasoningEffortModelId, isKimiK3ModelId, isMimoModelIdOrName, isMinimaxM2FamilyModelId, @@ -178,7 +179,8 @@ function fillThinkingWireDefaults( (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && supportsAdaptiveThinkingDisplay(spec.id); const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id); - const needsDefaultLevel = thinking.defaultLevel === undefined && isKimiK3ModelId(spec.id); + const needsDefaultLevel = + thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)); if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) { return thinking; } @@ -216,7 +218,7 @@ export function deriveThinking(spec: ModelSpec, compat: mode: inferThinkingControlMode(spec, parsed), efforts, }; - if (isKimiK3ModelId(spec.id)) { + if (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)) { config.defaultLevel = Effort.Max; } const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts); @@ -312,6 +314,13 @@ function getModelDefinedEfforts( spec: ModelSpec, compat: CompatOf, ): readonly Effort[] | undefined { + if (isGlm53ReasoningEffortModelId(spec.id)) { + // GLM-5.3+ exposes a uniform wire-exact low/high/max ladder on every + // host — unlike GLM-5.2, whose reasoning_effort dialect is + // host-specific. Thinking can no longer be disabled (handled by + // impliesMandatoryReasoning), and the default effort is `max`. + return LOW_HIGH_MAX_REASONING_EFFORTS; + } if (isGlm52ReasoningEffortModelId(spec.id)) { // GLM-5.2's reasoning_effort dialect is host-specific (verified against // live endpoints): @@ -572,6 +581,9 @@ function impliesMandatoryReasoning(parsed: ParsedModel, modelId: string): boolea if (parsed.kind === "pro" && semverGte(parsed.version, "2.5")) return true; } if (isKimiK3ModelId(modelId)) return true; + // GLM-5.3+ no longer supports disabling thinking — thinking.type must + // always be "enabled". Floor thinking-off requests to the lowest effort. + if (isGlm53ReasoningEffortModelId(modelId)) return true; if (isMinimaxM2FamilyModelId(modelId)) return true; if (OPENAI_O_SERIES_RE.test(bareModelId(modelId))) return true; return findThinkingVariantToken(modelId) !== undefined; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 5dc7d1f9b..6c62a3435 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -105719,6 +105719,35 @@ ] } }, + "glm-5.3": { + "id": "glm-5.3", + "name": "GLM-5.3", + "api": "anthropic-messages", + "provider": "zai", + "baseUrl": "https://api.z.ai/api/anthropic", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "anthropic-budget-effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "requiresEffort": true + } + }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index 20af095aa..97f7ef54d 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [ }, { id: "zai", - defaultModel: "glm-5.2", + defaultModel: "glm-5.3", envVars: ["ZAI_API_KEY"], createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config), catalogDiscovery: { label: "zAI" }, diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 3c88063ba..041973480 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -240,6 +240,40 @@ describe("generated model policies", () => { expect(models[0]?.maxTokens).toBe(131_072); }); + it("pins zai glm-5.3 to 1M context and derives uniform low/high/max thinking with mandatory reasoning", () => { + const models = [ + createSpec({ + id: "glm-5.3", + api: "anthropic-messages", + provider: "zai", + contextWindow: 200_000, + maxTokens: 8192, + }), + createSpec({ + id: "glm-5.3", + api: "openai-completions", + provider: "zhipu-coding-plan", + contextWindow: 200_000, + maxTokens: 8192, + }), + ]; + + applyGeneratedModelPolicies(models); + + // Context pinning — same 1M tier as glm-5.2 on both GLM coding-plan hosts. + for (const model of models) { + expect(model.contextWindow).toBe(1_000_000); + expect(model.maxTokens).toBe(131_072); + // Uniform wire-exact low/high/max ladder (NOT the host-specific + // high/max scale GLM-5.2 uses on zai/zhipu). + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]); + // Thinking can no longer be disabled. + expect(model.thinking?.requiresEffort).toBe(true); + // Default effort is `max` per the GLM-5.3 API spec. + expect(model.thinking?.defaultLevel).toBe(Effort.Max); + } + }); + it("pins MiniMax-M3 long-context providers to 1M context", () => { const models = [ createSpec({ diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index b69b9afb7..e6d82df3a 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -63,7 +63,7 @@ function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> { describe("zhipu-coding-plan descriptor", () => { it("defaults to the same Zhipu-hosted model used by login validation", () => { expect(DEFAULT_MODEL_PER_PROVIDER["zhipu-coding-plan"]).toBe("glm-5.1"); - expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.2"); + expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.3"); }); });