Files
oldschoola e49ee4b4e2 feat(catalog): add GLM-5.3 support with uniform low/high/max effort ladder and mandatory thinking
GLM-5.3 introduces three key API changes from GLM-5.2:
- Uniform wire-exact low/high/max reasoning_effort ladder on every host
  (replacing GLM-5.2's host-specific dialects)
- Thinking can no longer be disabled (thinking.type must always be "enabled")
- Default effort is max

Changes:
- Add isGlm53ReasoningEffortModelId classifier (>=5.3, base/air/turbo, non-vision)
- getModelDefinedEfforts: GLM-5.3 returns LOW_HIGH_MAX uniformly
- impliesMandatoryReasoning: GLM-5.3 floors thinking-off to lowest effort
- deriveThinking/fillThinkingWireDefaults: defaultLevel=max for GLM-5.3
- generated-policies: pin glm-5.3 to 1M context (zai + zhipu-coding-plan)
- descriptors: zai defaultModel -> glm-5.3
- generate-models: curated seed (glm-5.3 is live but not in /models discovery)
- models.json: bundled glm-5.3 entry
- Tests: catalog thinking-metadata + AI wire-mapping (5 new tests)
2026-08-13 23:18:17 -07:00

179 lines
6.6 KiB
TypeScript

import { describe, expect, it } from "bun:test";
import { buildOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { DEFAULT_MODEL_PER_PROVIDER } from "@oh-my-pi/pi-catalog/provider-models";
import { zhipuCodingPlanModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types";
/**
* Resolver-branch coverage for the `isZhipu` path added by the
* `zhipu-coding-plan` provider. GLM-5.2+ additionally accepts
* `reasoning_effort`; older BigModel thinking SKUs keep the binary Z.AI-shaped
* toggle only.
*/
const baseModel: Omit<ModelSpec<"openai-completions">, "provider" | "baseUrl"> = {
api: "openai-completions",
id: "glm-4.7",
name: "GLM-4.7",
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
maxTokens: 32_000,
contextWindow: 200_000,
reasoning: true,
};
function zhipuByProvider(): ModelSpec<"openai-completions"> {
return {
...baseModel,
provider: "zhipu-coding-plan",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
};
}
function zhipuByBaseUrl(): ModelSpec<"openai-completions"> {
return {
...baseModel,
// Provider intentionally not "zhipu-coding-plan" — exercises the
// URL-based fallback branch.
provider: "custom",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4/chat/completions",
};
}
function zhipuGlm52ByProvider(): ModelSpec<"openai-completions"> {
return {
...baseModel,
id: "glm-5.2",
name: "GLM-5.2",
provider: "zhipu-coding-plan",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
};
}
function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> {
return {
...baseModel,
id: "glm-5.2",
name: "GLM-5.2",
provider: "custom",
baseUrl: "https://open.bigmodel.cn/api/paas/v4",
};
}
describe("zhipu-coding-plan descriptor", () => {
it("defaults to the same Zhipu-hosted model used by login validation", () => {
expect(DEFAULT_MODEL_PER_PROVIDER["zhipu-coding-plan"]).toBe("glm-5.1");
expect(DEFAULT_MODEL_PER_PROVIDER.zai).toBe("glm-5.3");
});
});
describe("openai-completions compat — zhipu-coding-plan branch", () => {
it("forces zai thinking format and disables reasoning_effort before GLM-5.2", () => {
const compat = buildOpenAICompat(zhipuByProvider());
expect(compat.thinkingFormat).toBe("zai");
expect(compat.supportsReasoningEffort).toBe(false);
expect(compat.supportsDeveloperRole).toBe(false);
expect(compat.reasoningContentField).toBe("reasoning_content");
// Zhipu shares the multi-system-message tolerance of Z.AI.
expect(compat.supportsMultipleSystemMessages).toBe(true);
// `isZhipu` participates in the non-standard set, so `store` is off.
expect(compat.supportsStore).toBe(false);
});
it("detects zhipu by baseUrl when provider id is custom", () => {
const compat = buildOpenAICompat(zhipuByBaseUrl());
expect(compat.thinkingFormat).toBe("zai");
expect(compat.supportsReasoningEffort).toBe(false);
});
it("enables reasoning_effort for GLM-5.2 on both Zhipu route shapes", () => {
const codingPlanCompat = buildOpenAICompat(zhipuGlm52ByProvider());
const officialCompat = buildOpenAICompat(zhipuGlm52ByOfficialBaseUrl());
expect(codingPlanCompat.thinkingFormat).toBe("zai");
expect(codingPlanCompat.supportsReasoningEffort).toBe(true);
expect(officialCompat.thinkingFormat).toBe("zai");
expect(officialCompat.supportsReasoningEffort).toBe(true);
expect(officialCompat.reasoningContentField).toBe("reasoning_content");
expect(codingPlanCompat.maxTokensField).toBe("max_tokens");
expect(officialCompat.maxTokensField).toBe("max_tokens");
});
it("lets explicit model.compat overrides win at the resolver layer", () => {
const model: ModelSpec<"openai-completions"> = {
...zhipuByProvider(),
compat: {
supportsDeveloperRole: true,
supportsReasoningEffort: true,
thinkingFormat: "openai",
},
};
const resolved = buildOpenAICompat(model);
expect(resolved.supportsDeveloperRole).toBe(true);
expect(resolved.supportsReasoningEffort).toBe(true);
expect(resolved.thinkingFormat).toBe("openai");
// Untouched fields still come from the zhipu branch.
expect(resolved.reasoningContentField).toBe("reasoning_content");
});
});
describe("openai-completions compat — GLM coding-plan stream idle timeout", () => {
function glm52(provider: string, baseUrl: string): ModelSpec<"openai-completions"> {
return { ...baseModel, id: "glm-5.2", name: "GLM-5.2", provider, baseUrl };
}
// GLM coding-plan SKUs idle for minutes mid-reasoning; the 600s watchdog
// floor must apply on every gateway that fronts them, not just the native
// Z.AI/Zhipu hosts (issue #4758: GLM-5.2 via opencode-go stalled with
// "OpenAI completions stream stalled while waiting for the next event").
it("widens the idle timeout to 600s for GLM-5.x on Z.AI, Zhipu, and OpenCode gateways", () => {
expect(buildOpenAICompat(glm52("zai", "https://api.z.ai/api/coding/paas/v4")).streamIdleTimeoutMs).toBe(600_000);
expect(
buildOpenAICompat(glm52("zhipu-coding-plan", "https://open.bigmodel.cn/api/coding/paas/v4"))
.streamIdleTimeoutMs,
).toBe(600_000);
expect(buildOpenAICompat(glm52("opencode-go", "https://opencode.ai/zen/go/v1")).streamIdleTimeoutMs).toBe(
600_000,
);
expect(buildOpenAICompat(glm52("opencode-zen", "https://opencode.ai/zen/v1")).streamIdleTimeoutMs).toBe(600_000);
});
it("does not widen non-GLM models on the OpenCode gateway via the GLM floor", () => {
const kimi = buildOpenAICompat({
...baseModel,
id: "kimi-k2.5",
name: "Kimi K2.5",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
});
expect(kimi.streamIdleTimeoutMs).toBeUndefined();
});
});
describe("zhipu-coding-plan model discovery", () => {
it("uses the dedicated Coding Plan endpoint by default", async () => {
let requestedUrl = "";
const mockFetch: FetchImpl = Object.assign(
async (input: string | Request | URL): Promise<Response> => {
requestedUrl = input instanceof Request ? input.url : String(input);
return new Response(JSON.stringify({ data: [{ id: "glm-5.1", name: "GLM-5.1" }] }), {
headers: { "content-type": "application/json" },
});
},
{ preconnect: fetch.preconnect },
);
const options = zhipuCodingPlanModelManagerOptions({ apiKey: "test-key", fetch: mockFetch });
expect(typeof options.fetchDynamicModels).toBe("function");
expect(options.dynamicModelsAuthoritative).toBe(true);
const models = await options.fetchDynamicModels?.();
expect(requestedUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4/models");
expect(models?.[0]?.id).toBe("glm-5.1");
expect(models?.[0]?.baseUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4");
});
});