diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e398be7e3..c62cb414c 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Fixed + +- Fixed GLM-5.x coding-plan OpenAI-compatible streams to use a longer default watchdog window, avoiding spurious `OpenAI completions stream stalled while waiting for the next event` errors during slow `glm-5.1` thinking/output phases. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494)) +- Fixed `zhipu-coding-plan` model discovery and credential validation to use the dedicated GLM Coding Plan endpoint (`https://open.bigmodel.cn/api/coding/paas/v4`) instead of the general BigModel endpoint, preventing requests from consuming ordinary account balance. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494)) + ## [15.5.11] - 2026-05-29 ### Added diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index 7960d4cfb..54fd5c86f 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -870,7 +870,7 @@ export function zhipuCodingPlanModelManagerOptions( config?: ZhipuCodingPlanModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { const apiKey = config?.apiKey; - const baseUrl = config?.baseUrl ?? "https://open.bigmodel.cn/api/paas/v4"; + const baseUrl = config?.baseUrl ?? "https://open.bigmodel.cn/api/coding/paas/v4"; return { providerId: "zhipu-coding-plan", ...(apiKey && { @@ -2676,13 +2676,18 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe }, ), // --- Zhipu Coding Plan --- - openAiCompletionsDescriptor("zhipu-coding-plan", "zhipu-coding-plan", "https://open.bigmodel.cn/api/paas/v4", { - compat: { - thinkingFormat: "zai", - reasoningContentField: "reasoning_content", - supportsDeveloperRole: false, + openAiCompletionsDescriptor( + "zhipu-coding-plan", + "zhipu-coding-plan", + "https://open.bigmodel.cn/api/coding/paas/v4", + { + compat: { + thinkingFormat: "zai", + reasoningContentField: "reasoning_content", + supportsDeveloperRole: false, + }, }, - }), + ), ]; const filterActiveToolCallModels = (_id: string, m: ModelsDevModel): boolean => { diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 3ed4e78a4..b89c725a8 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -367,6 +367,25 @@ function getTrailingPartialDeepseekToken(text: string): string { const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE = "OpenAI completions stream timed out while waiting for the first event"; +const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; +const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; + +/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */ +export function getOpenAICompletionsStreamIdleTimeoutFallbackMs( + model: Model<"openai-completions">, +): number | undefined { + if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined; + if (model.provider === "zhipu-coding-plan" || model.provider === "zai") + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + + const baseUrl = model.baseUrl.toLowerCase(); + if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + } + + return undefined; +} + export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( model: Model<"openai-completions">, context: Context, @@ -387,7 +406,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( try { const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; - const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); + const idleTimeoutMs = + options?.streamIdleTimeoutMs ?? + getOpenAIStreamIdleTimeoutMs(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)); const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; diff --git a/packages/ai/src/utils/idle-iterator.ts b/packages/ai/src/utils/idle-iterator.ts index 7385612c9..0e20ba2bc 100644 --- a/packages/ai/src/utils/idle-iterator.ts +++ b/packages/ai/src/utils/idle-iterator.ts @@ -28,13 +28,14 @@ export function getStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_ /** * Returns the idle timeout used for OpenAI-family streaming transports. * + * `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` takes precedence over the generic + * `PI_STREAM_IDLE_TIMEOUT_MS` because some deployments tune OpenAI-compatible + * backends separately from Anthropic/Gemini-style transports. + * * Set `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0` to disable the watchdog. */ -export function getOpenAIStreamIdleTimeoutMs(): number | undefined { - return normalizeIdleTimeoutMs( - $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_STREAM_IDLE_TIMEOUT_MS, - DEFAULT_STREAM_IDLE_TIMEOUT_MS, - ); +export function getOpenAIStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_TIMEOUT_MS): number | undefined { + return normalizeIdleTimeoutMs($env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_STREAM_IDLE_TIMEOUT_MS, fallbackMs); } /** diff --git a/packages/ai/src/utils/oauth/zhipu.ts b/packages/ai/src/utils/oauth/zhipu.ts index 01035afac..14c766ed3 100644 --- a/packages/ai/src/utils/oauth/zhipu.ts +++ b/packages/ai/src/utils/oauth/zhipu.ts @@ -1,19 +1,19 @@ /** * Zhipu Coding Plan login flow. * - * Zhipu BigModel (智谱) provides an OpenAI-compatible API. - * API docs: https://docs.bigmodel.cn/cn/guide/develop/openai/introduction + * GLM Coding Plan provides an OpenAI-compatible API on the dedicated coding + * endpoint. API docs: https://docs.bigmodel.cn/cn/coding-plan/quick-start * * Simple API key flow: - * 1. User gets their API key from https://open.bigmodel.cn + * 1. User gets a Coding Plan API key from https://bigmodel.cn/coding-plan/personal/overview * 2. User pastes the API key into the CLI */ import { validateOpenAICompatibleApiKey } from "./api-key-validation"; import type { OAuthController } from "./types"; -const AUTH_URL = "https://open.bigmodel.cn/usercenter/apikeys"; -const API_BASE_URL = "https://open.bigmodel.cn/api/paas/v4"; +const AUTH_URL = "https://bigmodel.cn/coding-plan/personal/overview"; +const API_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4"; const VALIDATION_MODEL = "glm-5.1"; /** @@ -30,7 +30,7 @@ export async function loginZhipuCodingPlan(options: OAuthController): Promise { global.fetch = originalFetch; }); +describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { + it("widens GLM 5.1 coding-plan stream watchdogs", () => { + const model = { + ...openAICompletionsModel, + id: "glm-5.1", + name: "GLM-5.1", + provider: "zhipu-coding-plan", + baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4", + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); + }); + + it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => { + const model = { + ...openAICompletionsModel, + id: "glm-5.1", + name: "GLM-5.1", + provider: "openai", + baseUrl: "https://api.z.ai/api/coding/paas/v4", + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); + }); + + it("keeps ordinary OpenAI-compatible models on the global timeout", () => { + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined(); + }); +}); + /** * Contract: `isOpenAICompletionsProgressChunk` decides whether a streamed chunk * resets the idle-watchdog deadline in `iterateWithIdleTimeout`. A false diff --git a/packages/ai/test/stream-timeout-defaults.test.ts b/packages/ai/test/stream-timeout-defaults.test.ts index 35bbad1ab..c01beef75 100644 --- a/packages/ai/test/stream-timeout-defaults.test.ts +++ b/packages/ai/test/stream-timeout-defaults.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import { + getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout, @@ -56,6 +57,23 @@ describe("getStreamIdleTimeoutMs(fallbackMs)", () => { }); }); +describe("getOpenAIStreamIdleTimeoutMs(fallbackMs)", () => { + it("returns the per-provider fallback when OpenAI env vars are unset", () => { + expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBe(600_000); + }); + + it("lets PI_OPENAI_STREAM_IDLE_TIMEOUT_MS override the fallback before the generic env var", () => { + Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "42"; + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84"; + expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBe(84); + }); + + it("treats PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0 as a watchdog disable", () => { + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "0"; + expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBeUndefined(); + }); +}); + describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => { it("returns the per-provider fallback when env unset and idle timeout is undefined", () => { expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(300_000); diff --git a/packages/ai/test/zhipu-compat.test.ts b/packages/ai/test/zhipu-compat.test.ts index 02ddddd0d..ad45d53fd 100644 --- a/packages/ai/test/zhipu-compat.test.ts +++ b/packages/ai/test/zhipu-compat.test.ts @@ -1,4 +1,5 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, describe, expect, it } from "bun:test"; +import { zhipuCodingPlanModelManagerOptions } from "@oh-my-pi/pi-ai/provider-models/openai-compat"; import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat"; import type { Model } from "@oh-my-pi/pi-ai/types"; @@ -21,6 +22,11 @@ const baseModel: Omit, "provider" | "baseUrl"> = { reasoning: true, }; +const originalFetch = global.fetch; + +afterEach(() => { + global.fetch = originalFetch; +}); function zhipuByProvider(): Model<"openai-completions"> { return { ...baseModel, @@ -78,3 +84,24 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { expect(resolved.reasoningContentField).toBe("reasoning_content"); }); }); + +describe("zhipu-coding-plan model discovery", () => { + it("uses the dedicated Coding Plan endpoint by default", async () => { + let requestedUrl = ""; + const mockFetch = async (input: string | Request | URL): Promise => { + requestedUrl = input instanceof Request ? input.url : String(input); + return new Response(JSON.stringify({ data: [{ id: "glm-5.1", name: "GLM-5.1" }] }), { + headers: { "content-type": "application/json" }, + }); + }; + global.fetch = Object.assign(mockFetch, { preconnect: originalFetch.preconnect }); + + const options = zhipuCodingPlanModelManagerOptions({ apiKey: "test-key" }); + expect(typeof options.fetchDynamicModels).toBe("function"); + const models = await options.fetchDynamicModels?.(); + + expect(requestedUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4/models"); + expect(models?.[0]?.id).toBe("glm-5.1"); + expect(models?.[0]?.baseUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4"); + }); +});