feat(catalog): add GLM-5.3 support with uniform low/high/max effort ladder and mandatory thinking
GLM-5.3 introduces three key API changes from GLM-5.2: - Uniform wire-exact low/high/max reasoning_effort ladder on every host (replacing GLM-5.2's host-specific dialects) - Thinking can no longer be disabled (thinking.type must always be "enabled") - Default effort is max Changes: - Add isGlm53ReasoningEffortModelId classifier (>=5.3, base/air/turbo, non-vision) - getModelDefinedEfforts: GLM-5.3 returns LOW_HIGH_MAX uniformly - impliesMandatoryReasoning: GLM-5.3 floors thinking-off to lowest effort - deriveThinking/fillThinkingWireDefaults: defaultLevel=max for GLM-5.3 - generated-policies: pin glm-5.3 to 1M context (zai + zhipu-coding-plan) - descriptors: zai defaultModel -> glm-5.3 - generate-models: curated seed (glm-5.3 is live but not in /models discovery) - models.json: bundled glm-5.3 entry - Tests: catalog thinking-metadata + AI wire-mapping (5 new tests)
This commit is contained in:
@@ -555,6 +555,24 @@ async function generateModels() {
|
||||
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
|
||||
// applyAnthropicCatalogPolicy.
|
||||
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
|
||||
// Seed GLM-5.3 on the z.AI provider. GLM-5.3 is live on the Anthropic and
|
||||
// coding endpoints but not yet advertised in `/v1/models` (which still tops
|
||||
// out at glm-5.2), so endpoint discovery misses it. The zai provider is not
|
||||
// authoritative, so the seed survives regeneration; thinking metadata
|
||||
// (low/high/max uniform ladder, mandatory reasoning, defaultLevel=max) is
|
||||
// derived by rebakeModelThinking from the identity classifiers.
|
||||
allModels.push({
|
||||
id: "glm-5.3",
|
||||
name: "GLM-5.3",
|
||||
api: "anthropic-messages",
|
||||
provider: "zai",
|
||||
baseUrl: "https://api.z.ai/api/anthropic",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
} as ModelSpec<"anthropic-messages">);
|
||||
// Seed Meta's documented Muse model so first-run selection does not depend on
|
||||
// credentials or live discovery.
|
||||
allModels.push(...META_MUSE_STATIC_MODELS);
|
||||
|
||||
@@ -367,9 +367,13 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
model.omitMaxOutputTokens = true;
|
||||
}
|
||||
|
||||
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
|
||||
// GLM Coding Plan: the selectable 1M-context served ids; pin them so
|
||||
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
|
||||
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
|
||||
// GLM-5.3 succeeds GLM-5.2 with the same 1M context window.
|
||||
if (
|
||||
(model.provider === "zai" || model.provider === "zhipu-coding-plan") &&
|
||||
(model.id === "glm-5.2" || model.id === "glm-5.3")
|
||||
) {
|
||||
model.contextWindow = 1_000_000;
|
||||
model.maxTokens = 131_072;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user