diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index cb42cf1a3..15cd69791 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,9 +5,7 @@ ### Fixed - Fixed ChatGPT/Codex browser login missing connector OAuth scopes and rendering object-shaped token endpoint errors as `[object Object]`. ([#2825](https://github.com/can1357/oh-my-pi/issues/2825)) - -### Fixed - +- Fixed Zhipu/BigModel GLM-5.2 chat-completions requests so internal `xhigh` effort serializes as provider-native `reasoning_effort: "max"` and tool calls opt into `tool_stream`. ([#2833](https://github.com/can1357/oh-my-pi/issues/2833)) - Fixed Google Gemini CLI and Antigravity tool calls with `toolChoice: "auto"` serializing an explicit `toolConfig` AUTO mode, which can cause Gemini-3 models to leak raw planning JSON instead of executing tools. ([#2830](https://github.com/can1357/oh-my-pi/issues/2830)) ## [16.0.3] - 2026-06-16 diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index ee4e02fb2..395f1ae15 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1,6 +1,6 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; -import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity"; +import { isDeepseekModelIdOrName, isGlm52ReasoningEffortModelId } from "@oh-my-pi/pi-catalog/identity"; import { getSupportedEfforts, resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; @@ -367,7 +367,7 @@ export interface OpenAICompletionsOptions extends StreamOptions { openrouterVariant?: string; } -type OpenAICompletionsParams = ChatCompletionCreateParamsStreaming & { +type OpenAICompletionsParams = Omit & { top_k?: number; min_p?: number; repetition_penalty?: number; @@ -375,6 +375,8 @@ type OpenAICompletionsParams = ChatCompletionCreateParamsStreaming & { enable_thinking?: boolean; chat_template_kwargs?: { enable_thinking: boolean }; reasoning?: { effort?: string } | { enabled: false }; + reasoning_effort?: string | null; + tool_stream?: boolean; provider?: OpenAICompat["openRouterRouting"]; providerOptions?: { gateway?: { only?: string[]; order?: string[] } }; }; @@ -1338,6 +1340,10 @@ function buildParams( // `compat.alwaysSendMaxTokens` carries that detection. const requestedMaxTokens = options?.maxTokens ?? (compat.alwaysSendMaxTokens ? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS) : undefined); + const providerOutputClamp = + compat.thinkingFormat === "zai" && isGlm52ReasoningEffortModelId(model.id) + ? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS) + : OPENAI_MAX_OUTPUT_TOKENS; // OpenRouter fans out to upstreams whose output caps differ from the catalog // value (which tracks the highest-cap provider). A max_tokens above the routed // upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras @@ -1348,7 +1354,7 @@ function buildParams( const effectiveMaxTokens = requestedMaxTokens === undefined || omitMaxTokensForRouting ? undefined - : Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, OPENAI_MAX_OUTPUT_TOKENS); + : Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, providerOutputClamp); const requestModelId = resolveOpenAICompletionsModelId(model, options); const params: OpenAICompletionsParams = { @@ -1422,6 +1428,15 @@ function buildParams( // so LiteLLM → Bedrock never sees an empty `toolConfig` block. params.tools = []; } + if ( + compat.thinkingFormat === "zai" && + compat.supportsReasoningEffort && + isGlm52ReasoningEffortModelId(model.id) && + Array.isArray(params.tools) && + params.tools.length > 0 + ) { + params.tool_stream = true; + } if (options?.toolChoice && compat.supportsToolChoice) { params.tool_choice = mapToOpenAICompletionsToolChoice(options.toolChoice); @@ -1459,13 +1474,24 @@ function buildParams( } if (supportsReasoningParams && compat.thinkingFormat === "zai" && model.reasoning) { - // Z.ai uses binary thinking: { type: "enabled" | "disabled" } - // Must explicitly disable since z.ai defaults to thinking enabled. - const enabled = options?.reasoning && !options?.disableReasoning; + // Z.AI-style hosts use binary thinking, while GLM-5.2+ also accepts + // `reasoning_effort` when thinking is enabled. `minimal` maps to the + // provider's skip-thinking path, so keep the effort field absent there. + const requestedEffort = options?.reasoning; + const mappedEffort = + requestedEffort === undefined + ? undefined + : (compat.reasoningEffortMap?.[requestedEffort] ?? + model.thinking?.effortMap?.[requestedEffort] ?? + requestedEffort); + const enabled = mappedEffort !== undefined && mappedEffort !== "none" && !options?.disableReasoning; params.thinking = { type: enabled ? "enabled" : "disabled" }; if (enabled && compat.thinkingKeep) { params.thinking.keep = compat.thinkingKeep; } + if (enabled && compat.supportsReasoningEffort) { + params.reasoning_effort = mappedEffort; + } } else if (supportsReasoningParams && compat.thinkingFormat === "qwen" && model.reasoning) { // Qwen uses top-level enable_thinking: boolean params.enable_thinking = !!options?.reasoning && !options?.disableReasoning; diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index d5adcf8d8..4a4aea310 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -78,6 +78,26 @@ function baseContext(): Context { }; } +function zaiGlm52Model(): Model<"openai-completions"> { + return buildModel({ + id: "glm-5.2", + name: "GLM-5.2", + api: "openai-completions", + provider: "zhipu-coding-plan", + baseUrl: "https://open.bigmodel.cn/api/paas/v4", + reasoning: true, + compat: { + thinkingFormat: "zai", + reasoningContentField: "reasoning_content", + supportsDeveloperRole: false, + }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 131_072, + } satisfies ModelSpec<"openai-completions">); +} + async function captureOpenAICompletionsPayload( model: Model<"openai-completions">, context: Context = baseContext(), @@ -565,6 +585,56 @@ describe("openai-completions compatibility", () => { expect(getNestedBoolean(chatTemplateArgs, "enable_thinking")).toBe(true); }); + it("maps GLM-5.2 xhigh to Z.AI max and enables tool streaming", async () => { + const model = zaiGlm52Model(); + const readTool: Tool = { + name: "read", + description: "Read a file", + parameters: { + type: "object", + properties: { path: { type: "string" } }, + required: ["path"], + }, + }; + + const { promise, resolve } = Promise.withResolvers(); + streamOpenAICompletions( + model, + { ...baseContext(), tools: [readTool] }, + { + apiKey: "test-key", + reasoning: "xhigh", + signal: createAbortedSignal(), + onPayload: payload => resolve(payload), + maxTokens: 65_536, + }, + ); + const payload = await promise; + const thinking = getNestedObject(payload, "thinking"); + + expect(Reflect.get(thinking ?? {}, "type")).toBe("enabled"); + expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBe("max"); + expect(Reflect.get(toObject(payload) ?? {}, "tool_stream")).toBe(true); + expect(Reflect.get(toObject(payload) ?? {}, "max_tokens")).toBe(65_536); + }); + + it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => { + const model = zaiGlm52Model(); + + const { promise, resolve } = Promise.withResolvers(); + streamOpenAICompletions(model, baseContext(), { + apiKey: "test-key", + reasoning: "minimal", + signal: createAbortedSignal(), + onPayload: payload => resolve(payload), + }); + const payload = await promise; + const thinking = getNestedObject(payload, "thinking"); + + expect(Reflect.get(thinking ?? {}, "type")).toBe("disabled"); + expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBeUndefined(); + }); + it("treats finish_reason end as stop", async () => { const model: Model<"openai-completions"> = buildModel({ ...gpt4oMiniSpec, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 5c2321e0f..26e142f4a 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GLM-5.2 catalog thinking metadata for Zhipu/BigModel so the top effort is exposed as `xhigh` and maps to provider-native `max`. ([#2833](https://github.com/can1357/oh-my-pi/issues/2833)) + ## [16.0.2] - 2026-06-16 ### Fixed diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 330ff4ae4..e3402782e 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -208,9 +208,9 @@ function applyGeneratedModelPolicy(model: ModelSpec): void { model.maxTokens = copilotLimits.maxTokens; } - // GLM Coding Plan (zai): GLM-5.2 is the selectable 1M served id; pin it - // so endpoint discovery or older bundled fallbacks cannot regress to 200k. - if (model.provider === "zai" && model.id === "glm-5.2") { + // GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so + // endpoint discovery or older bundled fallbacks cannot regress to 200k. + if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") { model.contextWindow = 1_000_000; model.maxTokens = 131_072; } diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index a7265f448..3e2f3d72b 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -12,6 +12,7 @@ import { isAnthropicNamespacedModelId, isClaudeModelId, isDeepseekModelIdOrName, + isGlm52ReasoningEffortModelId, isKimiK26ModelId, isKimiModelId, isMimoModelIdOrName, @@ -82,6 +83,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isCerebras = modelMatchesHost(hostModel, "cerebras"); const isZai = modelMatchesHost(hostModel, "zai"); const isZhipu = modelMatchesHost(hostModel, "zhipu"); + const supportsZaiReasoningEffort = (isZai || isZhipu) && isGlm52ReasoningEffortModelId(spec.id); const isKilo = modelMatchesHost(hostModel, "kilo"); const isKimiModel = isKimiModelId(spec.id); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative"); @@ -136,6 +138,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const useMaxTokens = isMistral || isMoonshotNative || + isZai || + isZhipu || hostMatchesUrl(baseUrl, "chutes") || hostMatchesUrl(baseUrl, "fireworks") || isDirectDeepseekApi; @@ -202,7 +206,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // OpenAI's reasoning-API surface. supportsDeveloperRole: isOpenAIHost || isAzureHost, supportsMultipleSystemMessages: supportsMultipleSystemMessagesDefault, - supportsReasoningEffort: !isGrok && !isZai && !isZhipu && !isXiaomiMimo, + supportsReasoningEffort: !isGrok && !isXiaomiMimo && (!(isZai || isZhipu) || supportsZaiReasoningEffort), // GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale. supportsReasoningParams: provider !== "github-copilot", reasoningEffortMap: {}, diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index eb3fbf06b..a81c0d2e5 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -105,6 +105,17 @@ export function isReasoningGlmModelId(modelId: string): boolean { } return semverGte(glm.version, "4.5"); } +/** GLM-5.2+ coding SKUs accept `reasoning_effort` in addition to binary thinking. */ +export function isGlm52ReasoningEffortModelId(modelId: string): boolean { + const glm = parseGlmModel(bareModelId(modelId)); + if (!glm || glm.vision) { + return false; + } + if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") { + return false; + } + return semverGte(glm.version, "5.2"); +} /** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */ export function isGlmVisionModelId(modelId: string): boolean { diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index bcaaad94a..459e0ae93 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -23,6 +23,7 @@ import { import { findThinkingVariantToken, isDeepseekModelIdOrName, + isGlm52ReasoningEffortModelId, isMinimaxM2FamilyModelId, isOpenAIGptOssModelId, supportsAdaptiveThinkingDisplay, @@ -76,6 +77,13 @@ const DEEPSEEK_REASONING_EFFORT_MAP: Readonly = { const FIREWORKS_REASONING_EFFORT_MAP: Readonly = { [Effort.Minimal]: "none", }; +const ZAI_GLM_52_REASONING_EFFORT_MAP: Readonly = { + [Effort.Minimal]: "none", + [Effort.Low]: "high", + [Effort.Medium]: "high", + [Effort.High]: "high", + [Effort.XHigh]: "max", +}; /** * Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and @@ -259,11 +267,19 @@ function sameEffortList(left: readonly Effort[], right: readonly Effort[]): bool } function getModelDefinedEfforts(spec: ModelSpec): readonly Effort[] | undefined { + if (spec.api === "openai-completions" && isZaiGlm52ReasoningEffortModel(spec)) { + return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + } return spec.api === "openai-completions" && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id)) ? LOW_MEDIUM_HIGH_REASONING_EFFORTS : undefined; } +function isZaiGlm52ReasoningEffortModel(spec: ModelSpec): boolean { + if (!isGlm52ReasoningEffortModelId(spec.id)) return false; + return modelMatchesHost(spec, "zai") || modelMatchesHost(spec, "zhipu"); +} + function readCompatEffortMap(compat: CompatOf): EffortMap | undefined { if (compat === undefined || !("reasoningEffortMap" in compat)) { return undefined; @@ -288,6 +304,9 @@ function inferDetectedEffortMap( if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") { return GROQ_QWEN3_32B_REASONING_EFFORT_MAP; } + if (isZaiGlm52ReasoningEffortModel(spec)) { + return ZAI_GLM_52_REASONING_EFFORT_MAP; + } if (isDeepseekReasoningModel(spec)) { return DEEPSEEK_REASONING_EFFORT_MAP; } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 832140c9d..33ee99579 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -14801,6 +14801,38 @@ } } }, + "glm-5.2": { + "id": "glm-5.2", + "name": "GLM-5.2", + "api": "openai-completions", + "provider": "fireworks", + "baseUrl": "https://api.fireworks.ai/inference/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "none" + } + } + }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", @@ -27751,8 +27783,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 8192, + "contextWindow": 262144, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -27780,8 +27812,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 8192, + "contextWindow": 262144, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -30791,6 +30823,25 @@ ] } }, + "z-ai/glm-5.2": { + "id": "z-ai/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072 + }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM 5V Turbo", @@ -32875,6 +32926,35 @@ "high" ] } + }, + "kimi-k2.7-code-highspeed": { + "id": "kimi-k2.7-code-highspeed", + "name": "Kimi K2.7 Code HighSpeed", + "api": "openai-completions", + "provider": "moonshot", + "baseUrl": "https://api.moonshot.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.9, + "output": 8, + "cacheRead": 0.38, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } } }, "nanogpt": { @@ -41857,6 +41937,25 @@ ] } }, + "moonshotai/kimi-k2.7-code-highspeed": { + "id": "moonshotai/kimi-k2.7-code-highspeed", + "name": "moonshotai/kimi-k2.7-code-highspeed", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144 + }, "moonshotai/kimi-latest": { "id": "moonshotai/kimi-latest", "name": "Kimi Latest", @@ -46960,6 +47059,25 @@ "toolStrictMode": "mixed" } }, + "TEE/glm-5.2": { + "id": "TEE/glm-5.2", + "name": "TEE/glm-5.2", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072 + }, "TEE/gpt-oss-120b": { "id": "TEE/gpt-oss-120b", "name": "TEE/gpt-oss-120b", @@ -49461,6 +49579,35 @@ ] } }, + "zai-org/glm-5.2": { + "id": "zai-org/glm-5.2", + "name": "zai-org/glm-5.2", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "zai-org/glm-latest": { "id": "zai-org/glm-latest", "name": "GLM Latest", @@ -60012,8 +60159,8 @@ "text" ], "cost": { - "input": 0.098, - "output": 0.196, + "input": 0.09, + "output": 0.18, "cacheRead": 0.02, "cacheWrite": 0 }, @@ -62075,13 +62222,13 @@ "image" ], "cost": { - "input": 0.75, + "input": 0.74, "output": 3.5, - "cacheRead": 0.16, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -62397,8 +62544,8 @@ ], "cost": { "input": 0.5, - "output": 2.5, - "cacheRead": 0.15, + "output": 2.2, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -66658,13 +66805,13 @@ "text" ], "cost": { - "input": 0.125, + "input": 0.13, "output": 0.85, - "cacheRead": 0.06, + "cacheRead": 0.024999999999999998, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 131070, + "maxTokens": 98304, "thinking": { "mode": "effort", "efforts": [ @@ -66948,6 +67095,34 @@ ] } }, + "z-ai/glm-5.2": { + "id": "z-ai/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM 5V Turbo", @@ -67049,484 +67224,9 @@ } }, "synthetic": { - "hf:deepseek-ai/DeepSeek-R1": { - "id": "hf:deepseek-ai/DeepSeek-R1", - "name": "DeepSeek R1", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:deepseek-ai/DeepSeek-R1-0528": { - "id": "hf:deepseek-ai/DeepSeek-R1-0528", - "name": "DeepSeek R1 (0528)", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 8, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:deepseek-ai/DeepSeek-V3": { - "id": "hf:deepseek-ai/DeepSeek-V3", - "name": "DeepSeek V3", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1.25, - "output": 1.25, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:deepseek-ai/DeepSeek-V3-0324": { - "id": "hf:deepseek-ai/DeepSeek-V3-0324", - "name": "DeepSeek V3 (0324)", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1.2, - "output": 1.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000 - }, - "hf:deepseek-ai/DeepSeek-V3.1": { - "id": "hf:deepseek-ai/DeepSeek-V3.1", - "name": "DeepSeek V3.1", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.56, - "output": 1.68, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:deepseek-ai/DeepSeek-V3.1-Terminus": { - "id": "hf:deepseek-ai/DeepSeek-V3.1-Terminus", - "name": "DeepSeek V3.1 Terminus", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1.2, - "output": 1.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:deepseek-ai/DeepSeek-V3.2": { - "id": "hf:deepseek-ai/DeepSeek-V3.2", - "name": "DeepSeek V3.2", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.27, - "output": 0.4, - "cacheRead": 0.27, - "cacheWrite": 0 - }, - "contextWindow": 162816, - "maxTokens": 8000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } - }, - "hf:meta-llama/Llama-3.1-405B-Instruct": { - "id": "hf:meta-llama/Llama-3.1-405B-Instruct", - "name": "Llama-3.1-405B-Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 3, - "output": 3, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "hf:meta-llama/Llama-3.1-70B-Instruct": { - "id": "hf:meta-llama/Llama-3.1-70B-Instruct", - "name": "Llama-3.1-70B-Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.9, - "output": 0.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "hf:meta-llama/Llama-3.1-8B-Instruct": { - "id": "hf:meta-llama/Llama-3.1-8B-Instruct", - "name": "Llama-3.1-8B-Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.2, - "output": 0.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "hf:meta-llama/Llama-3.3-70B-Instruct": { - "id": "hf:meta-llama/Llama-3.3-70B-Instruct", - "name": "Llama-3.3-70B-Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.9, - "output": 0.9, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 128000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "hf:meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { - "id": "hf:meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", - "name": "Llama-4-Maverick-17B-128E-Instruct-FP8", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.22, - "output": 0.88, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 524000, - "maxTokens": 4096 - }, - "hf:meta-llama/Llama-4-Scout-17B-16E-Instruct": { - "id": "hf:meta-llama/Llama-4-Scout-17B-16E-Instruct", - "name": "Llama-4-Scout-17B-16E-Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.15, - "output": 0.6, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 328000, - "maxTokens": 4096 - }, - "hf:MiniMaxAI/MiniMax-M2": { - "id": "hf:MiniMaxAI/MiniMax-M2", - "name": "MiniMax-M2", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 196608, - "maxTokens": 131000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "hf:MiniMaxAI/MiniMax-M2.1": { - "id": "hf:MiniMaxAI/MiniMax-M2.1", - "name": "MiniMax-M2.1", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 204800, - "maxTokens": 131072, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "hf:MiniMaxAI/MiniMax-M2.5": { - "id": "hf:MiniMaxAI/MiniMax-M2.5", - "name": "MiniMax-M2.5", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.6, - "output": 3, - "cacheRead": 0.6, - "cacheWrite": 0 - }, - "contextWindow": 191488, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, "hf:MiniMaxAI/MiniMax-M3": { "id": "hf:MiniMaxAI/MiniMax-M3", - "name": "MiniMax-M3", + "name": "MiniMaxAI/MiniMax-M3", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67554,88 +67254,9 @@ ] } }, - "hf:moonshotai/Kimi-K2-Instruct-0905": { - "id": "hf:moonshotai/Kimi-K2-Instruct-0905", - "name": "Kimi K2 0905", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 1.2, - "output": 1.2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 32768 - }, - "hf:moonshotai/Kimi-K2-Thinking": { - "id": "hf:moonshotai/Kimi-K2-Thinking", - "name": "Kimi K2 Thinking", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 262144, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "requiresEffort": true - } - }, - "hf:moonshotai/Kimi-K2.5": { - "id": "hf:moonshotai/Kimi-K2.5", - "name": "Kimi K2.5", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:moonshotai/Kimi-K2.6": { "id": "hf:moonshotai/Kimi-K2.6", - "name": "Kimi K2.6", + "name": "moonshotai/Kimi-K2.6", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67663,39 +67284,9 @@ ] } }, - "hf:nvidia/Kimi-K2.5-NVFP4": { - "id": "hf:nvidia/Kimi-K2.5-NVFP4", - "name": "Kimi K2.5 (NVFP4)", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", - "name": "Nemotron 3 Super 120B", + "name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67724,7 +67315,7 @@ }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", - "name": "GPT OSS 120B", + "name": "openai/gpt-oss-120b", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67738,7 +67329,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 128000, + "contextWindow": 131072, "maxTokens": 32768, "thinking": { "mode": "effort", @@ -67749,76 +67340,9 @@ ] } }, - "hf:Qwen/Qwen3-235B-A22B-Instruct-2507": { - "id": "hf:Qwen/Qwen3-235B-A22B-Instruct-2507", - "name": "Qwen 3 235B Instruct", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0.2, - "output": 0.6, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 256000, - "maxTokens": 32000 - }, - "hf:Qwen/Qwen3-235B-A22B-Thinking-2507": { - "id": "hf:Qwen/Qwen3-235B-A22B-Thinking-2507", - "name": "Qwen3 235B A22B Thinking 2507", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.65, - "output": 3, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 256000, - "maxTokens": 32000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true - } - }, - "hf:Qwen/Qwen3-Coder-480B-A35B-Instruct": { - "id": "hf:Qwen/Qwen3-Coder-480B-A35B-Instruct", - "name": "Qwen 3 Coder 480B", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 2, - "output": 2, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 256000, - "maxTokens": 32000 - }, "hf:Qwen/Qwen3.5-397B-A17B": { "id": "hf:Qwen/Qwen3.5-397B-A17B", - "name": "Qwen3.5-97B-A17B", + "name": "Qwen/Qwen3.5-397B-A17B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67864,38 +67388,9 @@ "contextWindow": 262144, "maxTokens": 8192 }, - "hf:zai-org/GLM-4.6": { - "id": "hf:zai-org/GLM-4.6", - "name": "GLM 4.6", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:zai-org/GLM-4.7": { "id": "hf:zai-org/GLM-4.7", - "name": "GLM 4.7", + "name": "zai-org/GLM-4.7", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67909,7 +67404,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 200000, + "contextWindow": 202752, "maxTokens": 64000, "thinking": { "mode": "effort", @@ -67924,7 +67419,7 @@ }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", - "name": "GLM-4.7-Flash", + "name": "zai-org/GLM-4.7-Flash", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -67951,38 +67446,9 @@ ] } }, - "hf:zai-org/GLM-5": { - "id": "hf:zai-org/GLM-5", - "name": "GLM-5", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1, - "output": 3, - "cacheRead": 1, - "cacheWrite": 0 - }, - "contextWindow": 196608, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:zai-org/GLM-5.1": { "id": "hf:zai-org/GLM-5.1", - "name": "GLM 5.1", + "name": "zai-org/GLM-5.1", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -68791,8 +68257,8 @@ "text" ], "cost": { - "input": 2.5, - "output": 7.5, + "input": 1.25, + "output": 3.75, "cacheRead": 0, "cacheWrite": 0 }, @@ -68915,7 +68381,8 @@ "medium", "high", "xhigh" - ] + ], + "requiresEffort": true }, "compat": { "escapeBuiltinToolNames": true @@ -68972,7 +68439,7 @@ "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 131072, + "maxTokens": 131071, "thinking": { "mode": "budget", "efforts": [ @@ -68987,6 +68454,39 @@ "escapeBuiltinToolNames": true } }, + "umans-glm-5.2": { + "id": "umans-glm-5.2", + "name": "Umans GLM 5.2", + "api": "anthropic-messages", + "provider": "umans", + "baseUrl": "https://api.code.umans.ai", + "reasoning": true, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 405504, + "maxTokens": 131071, + "compat": { + "escapeBuiltinToolNames": true + } + }, "umans-kimi-k2.6": { "id": "umans-kimi-k2.6", "name": "Umans Kimi K2.6", @@ -69791,6 +69291,28 @@ "supportsUsageInStreaming": false } }, + "e2ee-glm-5-2-p": { + "id": "e2ee-glm-5-2-p", + "name": "e2ee-glm-5-2-p", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null, + "compat": { + "supportsUsageInStreaming": false + } + }, "e2ee-gpt-oss-120b-p": { "id": "e2ee-gpt-oss-120b-p", "name": "e2ee-gpt-oss-120b-p", @@ -70559,24 +70081,32 @@ }, "kimi-k2-7-code": { "id": "kimi-k2-7-code", - "name": "kimi-k2-7-code", + "name": "Kimi K2.7 Code", "api": "openai-completions", "provider": "venice", "baseUrl": "https://api.venice.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.9, + "output": 4.3, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 256000, "maxTokens": 32768, - "compat": { - "supportsUsageInStreaming": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] } }, "kimi-k2-thinking": { @@ -71996,6 +71526,35 @@ "xhigh" ] } + }, + "zai-org-glm-5-2": { + "id": "zai-org-glm-5-2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.75, + "output": 5.5, + "cacheRead": 0.325, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 24000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } } }, "vercel-ai-gateway": { @@ -74758,6 +74317,36 @@ ] } }, + "moonshotai/kimi-k2.7-code-highspeed": { + "id": "moonshotai/kimi-k2.7-code-highspeed", + "name": "Kimi K2.7 Code High Speed", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.9, + "output": 8, + "cacheRead": 0.38, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super", @@ -77063,6 +76652,35 @@ ] } }, + "zai/glm-5.2": { + "id": "zai/glm-5.2", + "name": "GLM 5.2", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "zai/glm-5v-turbo": { "id": "zai/glm-5v-turbo", "name": "GLM 5V Turbo", @@ -80593,6 +80211,25 @@ "requiresEffort": true } }, + "google/gemini-embedding-2": { + "id": "google/gemini-embedding-2", + "name": "Gemini Embedding 2", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": null + }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", "name": "Gemma 3 12B", @@ -81433,6 +81070,36 @@ ] } }, + "moonshotai/kimi-k2.7-code-highspeed": { + "id": "moonshotai/kimi-k2.7-code-highspeed", + "name": "Kimi K2.7 Code HighSpeed", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.9, + "output": 8, + "cacheRead": 0.38, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai/chat-latest": { "id": "openai/chat-latest", "name": "Chat Latest (GPT-5.5 Instant)", @@ -82479,6 +82146,25 @@ ] } }, + "qwen/qwen3-vl-embedding": { + "id": "qwen/qwen3-vl-embedding", + "name": "Qwen3 VL Embedding", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": null + }, "qwen/qwen3-vl-plus": { "id": "qwen/qwen3-vl-plus", "name": "Qwen3-VL-Plus", @@ -84000,6 +83686,44 @@ ] } }, + "z-ai/glm-5.2": { + "id": "z-ai/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072 + }, + "z-ai/glm-5.2-free": { + "id": "z-ai/glm-5.2-free", + "name": "GLM 5.2 (Free)", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072 + }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM 5V Turbo", @@ -84227,8 +83951,16 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "none", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } } }, "glm-5v-turbo": { diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 3ea680f7c..aa6f23714 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -477,6 +477,29 @@ describe("model thinking runtime helpers", () => { ); }); + it("maps GLM-5.2 xhigh to Z.AI provider-native max", () => { + const model = createModel({ + id: "glm-5.2", + api: "openai-completions", + provider: "zhipu-coding-plan", + baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4", + compat: { thinkingFormat: "zai" }, + }); + + expect(model.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "none", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }, + }); + expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh); + }); + it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => { const model = createModel({ id: "qwen/qwen3-32b", diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index 9d53ba5c5..fe575221a 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -5,10 +5,9 @@ import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; /** * Resolver-branch coverage for the `isZhipu` path added by the - * `zhipu-coding-plan` provider. Mirrors the shape of existing zai/cerebras - * tests: assert the contract the provider relies on (zai thinking format, - * disabled `reasoning_effort`, no `developer` role) so future refactors of - * `buildOpenAICompat` cannot silently regress the BigModel SKU. + * `zhipu-coding-plan` provider. GLM-5.2+ additionally accepts + * `reasoning_effort`; older BigModel thinking SKUs keep the binary Z.AI-shaped + * toggle only. */ const baseModel: Omit, "provider" | "baseUrl"> = { @@ -40,8 +39,28 @@ function zhipuByBaseUrl(): ModelSpec<"openai-completions"> { }; } +function zhipuGlm52ByProvider(): ModelSpec<"openai-completions"> { + return { + ...baseModel, + id: "glm-5.2", + name: "GLM-5.2", + provider: "zhipu-coding-plan", + baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4", + }; +} + +function zhipuGlm52ByOfficialBaseUrl(): ModelSpec<"openai-completions"> { + return { + ...baseModel, + id: "glm-5.2", + name: "GLM-5.2", + provider: "custom", + baseUrl: "https://open.bigmodel.cn/api/paas/v4", + }; +} + describe("openai-completions compat — zhipu-coding-plan branch", () => { - it("forces zai thinking format and disables reasoning_effort / developer role", () => { + it("forces zai thinking format and disables reasoning_effort before GLM-5.2", () => { const compat = buildOpenAICompat(zhipuByProvider()); expect(compat.thinkingFormat).toBe("zai"); @@ -61,6 +80,19 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { expect(compat.supportsReasoningEffort).toBe(false); }); + it("enables reasoning_effort for GLM-5.2 on both Zhipu route shapes", () => { + const codingPlanCompat = buildOpenAICompat(zhipuGlm52ByProvider()); + const officialCompat = buildOpenAICompat(zhipuGlm52ByOfficialBaseUrl()); + + expect(codingPlanCompat.thinkingFormat).toBe("zai"); + expect(codingPlanCompat.supportsReasoningEffort).toBe(true); + expect(officialCompat.thinkingFormat).toBe("zai"); + expect(officialCompat.supportsReasoningEffort).toBe(true); + expect(officialCompat.reasoningContentField).toBe("reasoning_content"); + expect(codingPlanCompat.maxTokensField).toBe("max_tokens"); + expect(officialCompat.maxTokensField).toBe("max_tokens"); + }); + it("lets explicit model.compat overrides win at the resolver layer", () => { const model: ModelSpec<"openai-completions"> = { ...zhipuByProvider(), diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f6c21a2e6..6c3512e05 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,10 +5,8 @@ ### Fixed - Fixed RPC/ACP startup forcing todo settings back to host defaults, so project-level `todo.enabled`, `todo.reminders`, and `todo.eager` opt-outs now suppress protocol-mode todo prompt injection; enabled todo reminders are now persisted to the JSONL transcript so the log matches the model-visible context ([#2824](https://github.com/can1357/oh-my-pi/issues/2824)). - -### Fixed - - Fixed default prompts to instruct the agent to read applicable `skill://` content before starting work, so discovered skills influence broad task requests like frontend generation ([#2829](https://github.com/can1357/oh-my-pi/issues/2829)). +- Fixed hashline visible-line validation for ACP editor reads so `INS.POST` anchors displayed by bridge-backed range and multi-range `read` output are merged into the session snapshot before `edit` validates them ([#2773](https://github.com/can1357/oh-my-pi/issues/2773)). ## [16.0.3] - 2026-06-16 @@ -80,10 +78,6 @@ - Fixed task subagents to install their configured ordered model candidates as child-session retry fallback chains, so retryable provider failures can advance to the next subagent model instead of failing the worker ([#2750](https://github.com/can1357/oh-my-pi/issues/2750)). - Fixed empty reasonless aborted assistant turns to auto-retry without switching model fallback, so transient provider-side aborts after tool results do not end headless sessions ([#2685](https://github.com/can1357/oh-my-pi/issues/2685)). -### Fixed - -- Fixed hashline visible-line validation for ACP editor reads so `INS.POST` anchors displayed by bridge-backed range and multi-range `read` output are merged into the session snapshot before `edit` validates them ([#2773](https://github.com/can1357/oh-my-pi/issues/2773)). - ## [16.0.1] - 2026-06-15 ### Breaking Changes