feat(catalog): added GLM-5.2 reasoning support for ZAI and zhipu completions

- Added ZAI GLM-5.2 reasoning-effort mapping, translating minimal to none and xhigh to max.
- Enabled ZAI and zhipu GLM-5.2 completion requests to send reasoning_effort and tool_stream.
- Added provider token clamping so GLM-5.2 completion requests use capped max_tokens.
- Updated catalog policies to route GLM-5.2 max-token and reasoning support through ZAI/zhipu hosts.
- Removed synthetic HF model entries and aligned GLM-5.2 catalog specs with real providers.

Fixes #2833
This commit is contained in:
can1357
2026-06-17 09:37:44 +02:00
parent 22bd89912d
commit fb3534740f
12 changed files with 685 additions and 772 deletions
+1 -3
View File
@@ -5,9 +5,7 @@
### Fixed
- Fixed ChatGPT/Codex browser login missing connector OAuth scopes and rendering object-shaped token endpoint errors as `[object Object]`. ([#2825](https://github.com/can1357/oh-my-pi/issues/2825))
### Fixed
- Fixed Zhipu/BigModel GLM-5.2 chat-completions requests so internal `xhigh` effort serializes as provider-native `reasoning_effort: "max"` and tool calls opt into `tool_stream`. ([#2833](https://github.com/can1357/oh-my-pi/issues/2833))
- Fixed Google Gemini CLI and Antigravity tool calls with `toolChoice: "auto"` serializing an explicit `toolConfig` AUTO mode, which can cause Gemini-3 models to leak raw planning JSON instead of executing tools. ([#2830](https://github.com/can1357/oh-my-pi/issues/2830))
## [16.0.3] - 2026-06-16
@@ -1,6 +1,6 @@
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity";
import { isDeepseekModelIdOrName, isGlm52ReasoningEffortModelId } from "@oh-my-pi/pi-catalog/identity";
import { getSupportedEfforts, resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
@@ -367,7 +367,7 @@ export interface OpenAICompletionsOptions extends StreamOptions {
openrouterVariant?: string;
}
type OpenAICompletionsParams = ChatCompletionCreateParamsStreaming & {
type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming, "reasoning_effort"> & {
top_k?: number;
min_p?: number;
repetition_penalty?: number;
@@ -375,6 +375,8 @@ type OpenAICompletionsParams = ChatCompletionCreateParamsStreaming & {
enable_thinking?: boolean;
chat_template_kwargs?: { enable_thinking: boolean };
reasoning?: { effort?: string } | { enabled: false };
reasoning_effort?: string | null;
tool_stream?: boolean;
provider?: OpenAICompat["openRouterRouting"];
providerOptions?: { gateway?: { only?: string[]; order?: string[] } };
};
@@ -1338,6 +1340,10 @@ function buildParams(
// `compat.alwaysSendMaxTokens` carries that detection.
const requestedMaxTokens =
options?.maxTokens ?? (compat.alwaysSendMaxTokens ? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS) : undefined);
const providerOutputClamp =
compat.thinkingFormat === "zai" && isGlm52ReasoningEffortModelId(model.id)
? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS)
: OPENAI_MAX_OUTPUT_TOKENS;
// OpenRouter fans out to upstreams whose output caps differ from the catalog
// value (which tracks the highest-cap provider). A max_tokens above the routed
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
@@ -1348,7 +1354,7 @@ function buildParams(
const effectiveMaxTokens =
requestedMaxTokens === undefined || omitMaxTokensForRouting
? undefined
: Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, OPENAI_MAX_OUTPUT_TOKENS);
: Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, providerOutputClamp);
const requestModelId = resolveOpenAICompletionsModelId(model, options);
const params: OpenAICompletionsParams = {
@@ -1422,6 +1428,15 @@ function buildParams(
// so LiteLLM → Bedrock never sees an empty `toolConfig` block.
params.tools = [];
}
if (
compat.thinkingFormat === "zai" &&
compat.supportsReasoningEffort &&
isGlm52ReasoningEffortModelId(model.id) &&
Array.isArray(params.tools) &&
params.tools.length > 0
) {
params.tool_stream = true;
}
if (options?.toolChoice && compat.supportsToolChoice) {
params.tool_choice = mapToOpenAICompletionsToolChoice(options.toolChoice);
@@ -1459,13 +1474,24 @@ function buildParams(
}
if (supportsReasoningParams && compat.thinkingFormat === "zai" && model.reasoning) {
// Z.ai uses binary thinking: { type: "enabled" | "disabled" }
// Must explicitly disable since z.ai defaults to thinking enabled.
const enabled = options?.reasoning && !options?.disableReasoning;
// Z.AI-style hosts use binary thinking, while GLM-5.2+ also accepts
// `reasoning_effort` when thinking is enabled. `minimal` maps to the
// provider's skip-thinking path, so keep the effort field absent there.
const requestedEffort = options?.reasoning;
const mappedEffort =
requestedEffort === undefined
? undefined
: (compat.reasoningEffortMap?.[requestedEffort] ??
model.thinking?.effortMap?.[requestedEffort] ??
requestedEffort);
const enabled = mappedEffort !== undefined && mappedEffort !== "none" && !options?.disableReasoning;
params.thinking = { type: enabled ? "enabled" : "disabled" };
if (enabled && compat.thinkingKeep) {
params.thinking.keep = compat.thinkingKeep;
}
if (enabled && compat.supportsReasoningEffort) {
params.reasoning_effort = mappedEffort;
}
} else if (supportsReasoningParams && compat.thinkingFormat === "qwen" && model.reasoning) {
// Qwen uses top-level enable_thinking: boolean
params.enable_thinking = !!options?.reasoning && !options?.disableReasoning;
@@ -78,6 +78,26 @@ function baseContext(): Context {
};
}
function zaiGlm52Model(): Model<"openai-completions"> {
return buildModel({
id: "glm-5.2",
name: "GLM-5.2",
api: "openai-completions",
provider: "zhipu-coding-plan",
baseUrl: "https://open.bigmodel.cn/api/paas/v4",
reasoning: true,
compat: {
thinkingFormat: "zai",
reasoningContentField: "reasoning_content",
supportsDeveloperRole: false,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"openai-completions">);
}
async function captureOpenAICompletionsPayload(
model: Model<"openai-completions">,
context: Context = baseContext(),
@@ -565,6 +585,56 @@ describe("openai-completions compatibility", () => {
expect(getNestedBoolean(chatTemplateArgs, "enable_thinking")).toBe(true);
});
it("maps GLM-5.2 xhigh to Z.AI max and enables tool streaming", async () => {
const model = zaiGlm52Model();
const readTool: Tool = {
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" } },
required: ["path"],
},
};
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(
model,
{ ...baseContext(), tools: [readTool] },
{
apiKey: "test-key",
reasoning: "xhigh",
signal: createAbortedSignal(),
onPayload: payload => resolve(payload),
maxTokens: 65_536,
},
);
const payload = await promise;
const thinking = getNestedObject(payload, "thinking");
expect(Reflect.get(thinking ?? {}, "type")).toBe("enabled");
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBe("max");
expect(Reflect.get(toObject(payload) ?? {}, "tool_stream")).toBe(true);
expect(Reflect.get(toObject(payload) ?? {}, "max_tokens")).toBe(65_536);
});
it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => {
const model = zaiGlm52Model();
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, baseContext(), {
apiKey: "test-key",
reasoning: "minimal",
signal: createAbortedSignal(),
onPayload: payload => resolve(payload),
});
const payload = await promise;
const thinking = getNestedObject(payload, "thinking");
expect(Reflect.get(thinking ?? {}, "type")).toBe("disabled");
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBeUndefined();
});
it("treats finish_reason end as stop", async () => {
const model: Model<"openai-completions"> = buildModel({
...gpt4oMiniSpec,
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed GLM-5.2 catalog thinking metadata for Zhipu/BigModel so the top effort is exposed as `xhigh` and maps to provider-native `max`. ([#2833](https://github.com/can1357/oh-my-pi/issues/2833))
## [16.0.2] - 2026-06-16
### Fixed
@@ -208,9 +208,9 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.maxTokens = copilotLimits.maxTokens;
}
// GLM Coding Plan (zai): GLM-5.2 is the selectable 1M served id; pin it
// so endpoint discovery or older bundled fallbacks cannot regress to 200k.
if (model.provider === "zai" && model.id === "glm-5.2") {
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
model.contextWindow = 1_000_000;
model.maxTokens = 131_072;
}
+5 -1
View File
@@ -12,6 +12,7 @@ import {
isAnthropicNamespacedModelId,
isClaudeModelId,
isDeepseekModelIdOrName,
isGlm52ReasoningEffortModelId,
isKimiK26ModelId,
isKimiModelId,
isMimoModelIdOrName,
@@ -82,6 +83,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
const isCerebras = modelMatchesHost(hostModel, "cerebras");
const isZai = modelMatchesHost(hostModel, "zai");
const isZhipu = modelMatchesHost(hostModel, "zhipu");
const supportsZaiReasoningEffort = (isZai || isZhipu) && isGlm52ReasoningEffortModelId(spec.id);
const isKilo = modelMatchesHost(hostModel, "kilo");
const isKimiModel = isKimiModelId(spec.id);
const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative");
@@ -136,6 +138,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
const useMaxTokens =
isMistral ||
isMoonshotNative ||
isZai ||
isZhipu ||
hostMatchesUrl(baseUrl, "chutes") ||
hostMatchesUrl(baseUrl, "fireworks") ||
isDirectDeepseekApi;
@@ -202,7 +206,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// OpenAI's reasoning-API surface.
supportsDeveloperRole: isOpenAIHost || isAzureHost,
supportsMultipleSystemMessages: supportsMultipleSystemMessagesDefault,
supportsReasoningEffort: !isGrok && !isZai && !isZhipu && !isXiaomiMimo,
supportsReasoningEffort: !isGrok && !isXiaomiMimo && (!(isZai || isZhipu) || supportsZaiReasoningEffort),
// GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale.
supportsReasoningParams: provider !== "github-copilot",
reasoningEffortMap: {},
+11
View File
@@ -105,6 +105,17 @@ export function isReasoningGlmModelId(modelId: string): boolean {
}
return semverGte(glm.version, "4.5");
}
/** GLM-5.2+ coding SKUs accept `reasoning_effort` in addition to binary thinking. */
export function isGlm52ReasoningEffortModelId(modelId: string): boolean {
const glm = parseGlmModel(bareModelId(modelId));
if (!glm || glm.vision) {
return false;
}
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
return false;
}
return semverGte(glm.version, "5.2");
}
/** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */
export function isGlmVisionModelId(modelId: string): boolean {
+19
View File
@@ -23,6 +23,7 @@ import {
import {
findThinkingVariantToken,
isDeepseekModelIdOrName,
isGlm52ReasoningEffortModelId,
isMinimaxM2FamilyModelId,
isOpenAIGptOssModelId,
supportsAdaptiveThinkingDisplay,
@@ -76,6 +77,13 @@ const DEEPSEEK_REASONING_EFFORT_MAP: Readonly<EffortMap> = {
const FIREWORKS_REASONING_EFFORT_MAP: Readonly<EffortMap> = {
[Effort.Minimal]: "none",
};
const ZAI_GLM_52_REASONING_EFFORT_MAP: Readonly<EffortMap> = {
[Effort.Minimal]: "none",
[Effort.Low]: "high",
[Effort.Medium]: "high",
[Effort.High]: "high",
[Effort.XHigh]: "max",
};
/**
* Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and
@@ -259,11 +267,19 @@ function sameEffortList(left: readonly Effort[], right: readonly Effort[]): bool
}
function getModelDefinedEfforts<TApi extends Api>(spec: ModelSpec<TApi>): readonly Effort[] | undefined {
if (spec.api === "openai-completions" && isZaiGlm52ReasoningEffortModel(spec)) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return spec.api === "openai-completions" && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id))
? LOW_MEDIUM_HIGH_REASONING_EFFORTS
: undefined;
}
function isZaiGlm52ReasoningEffortModel<TApi extends Api>(spec: ModelSpec<TApi>): boolean {
if (!isGlm52ReasoningEffortModelId(spec.id)) return false;
return modelMatchesHost(spec, "zai") || modelMatchesHost(spec, "zhipu");
}
function readCompatEffortMap(compat: CompatOf<Api>): EffortMap | undefined {
if (compat === undefined || !("reasoningEffortMap" in compat)) {
return undefined;
@@ -288,6 +304,9 @@ function inferDetectedEffortMap<TApi extends Api>(
if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") {
return GROQ_QWEN3_32B_REASONING_EFFORT_MAP;
}
if (isZaiGlm52ReasoningEffortModel(spec)) {
return ZAI_GLM_52_REASONING_EFFORT_MAP;
}
if (isDeepseekReasoningModel(spec)) {
return DEEPSEEK_REASONING_EFFORT_MAP;
}
File diff suppressed because it is too large Load Diff
@@ -477,6 +477,29 @@ describe("model thinking runtime helpers", () => {
);
});
it("maps GLM-5.2 xhigh to Z.AI provider-native max", () => {
const model = createModel({
id: "glm-5.2",
api: "openai-completions",
provider: "zhipu-coding-plan",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
compat: { thinkingFormat: "zai" },
});
expect(model.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
effortMap: {
minimal: "none",
low: "high",
medium: "high",
high: "high",
xhigh: "max",
},
});
expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh);
});
it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => {
const model = createModel({
id: "qwen/qwen3-32b",
+37 -5
View File
@@ -5,10 +5,9 @@ import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types";
/**
* Resolver-branch coverage for the `isZhipu` path added by the
* `zhipu-coding-plan` provider. Mirrors the shape of existing zai/cerebras
* tests: assert the contract the provider relies on (zai thinking format,
* disabled `reasoning_effort`, no `developer` role) so future refactors of
* `buildOpenAICompat` cannot silently regress the BigModel SKU.
* `zhipu-coding-plan` provider. GLM-5.2+ additionally accepts
* `reasoning_effort`; older BigModel thinking SKUs keep the binary Z.AI-shaped
* toggle only.
*/
const baseModel: Omit<ModelSpec<"openai-completions">, "provider" | "baseUrl"> = {
@@ -40,8 +39,28 @@ function zhipuByBaseUrl(): ModelSpec<"openai-completions"> {
};
}
function zhipuGlm52ByProvider(): ModelSpec<"openai-completions"> {
return {
...baseModel,
id: "glm-5.2",
name: "GLM-5.2",
provider: "zhipu-coding-plan",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
};
}
function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> {
return {
...baseModel,
id: "glm-5.2",
name: "GLM-5.2",
provider: "custom",
baseUrl: "https://open.bigmodel.cn/api/paas/v4",
};
}
describe("openai-completions compat — zhipu-coding-plan branch", () => {
it("forces zai thinking format and disables reasoning_effort / developer role", () => {
it("forces zai thinking format and disables reasoning_effort before GLM-5.2", () => {
const compat = buildOpenAICompat(zhipuByProvider());
expect(compat.thinkingFormat).toBe("zai");
@@ -61,6 +80,19 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => {
expect(compat.supportsReasoningEffort).toBe(false);
});
it("enables reasoning_effort for GLM-5.2 on both Zhipu route shapes", () => {
const codingPlanCompat = buildOpenAICompat(zhipuGlm52ByProvider());
const officialCompat = buildOpenAICompat(zhipuGlm52ByOfficialBaseUrl());
expect(codingPlanCompat.thinkingFormat).toBe("zai");
expect(codingPlanCompat.supportsReasoningEffort).toBe(true);
expect(officialCompat.thinkingFormat).toBe("zai");
expect(officialCompat.supportsReasoningEffort).toBe(true);
expect(officialCompat.reasoningContentField).toBe("reasoning_content");
expect(codingPlanCompat.maxTokensField).toBe("max_tokens");
expect(officialCompat.maxTokensField).toBe("max_tokens");
});
it("lets explicit model.compat overrides win at the resolver layer", () => {
const model: ModelSpec<"openai-completions"> = {
...zhipuByProvider(),
+1 -7
View File
@@ -5,10 +5,8 @@
### Fixed
- Fixed RPC/ACP startup forcing todo settings back to host defaults, so project-level `todo.enabled`, `todo.reminders`, and `todo.eager` opt-outs now suppress protocol-mode todo prompt injection; enabled todo reminders are now persisted to the JSONL transcript so the log matches the model-visible context ([#2824](https://github.com/can1357/oh-my-pi/issues/2824)).
### Fixed
- Fixed default prompts to instruct the agent to read applicable `skill://<name>` content before starting work, so discovered skills influence broad task requests like frontend generation ([#2829](https://github.com/can1357/oh-my-pi/issues/2829)).
- Fixed hashline visible-line validation for ACP editor reads so `INS.POST` anchors displayed by bridge-backed range and multi-range `read` output are merged into the session snapshot before `edit` validates them ([#2773](https://github.com/can1357/oh-my-pi/issues/2773)).
## [16.0.3] - 2026-06-16
@@ -80,10 +78,6 @@
- Fixed task subagents to install their configured ordered model candidates as child-session retry fallback chains, so retryable provider failures can advance to the next subagent model instead of failing the worker ([#2750](https://github.com/can1357/oh-my-pi/issues/2750)).
- Fixed empty reasonless aborted assistant turns to auto-retry without switching model fallback, so transient provider-side aborts after tool results do not end headless sessions ([#2685](https://github.com/can1357/oh-my-pi/issues/2685)).
### Fixed
- Fixed hashline visible-line validation for ACP editor reads so `INS.POST` anchors displayed by bridge-backed range and multi-range `read` output are merged into the session snapshot before `edit` validates them ([#2773](https://github.com/can1357/oh-my-pi/issues/2773)).
## [16.0.1] - 2026-06-15
### Breaking Changes