From 7cdecd824a871efce161fe68aa9e230ca8a8788b Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 15:56:49 +0000 Subject: [PATCH] fix(catalog): widened GLM coding-plan idle timeout to opencode gateways GLM-5.x coding-plan SKUs idle for minutes mid-reasoning, so they get a 600s stream idle-timeout floor instead of the 120s default. That floor was gated to the native Z.AI/Zhipu hosts only, so GLM-5.2 served through the OpenCode Go/Zen gateways fell back to the 120s watchdog and stalled with "OpenAI completions stream stalled while waiting for the next event" during the slow /plan writing phase. Fixes #4758 --- packages/catalog/CHANGELOG.md | 4 +++ packages/catalog/src/compat/openai.ts | 2 +- packages/catalog/test/zhipu-compat.test.ts | 33 ++++++++++++++++++++++ 3 files changed, 38 insertions(+), 1 deletion(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..ee14f3ef7 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GLM-5.x coding-plan streams via the OpenCode Go/Zen gateways (`opencode.ai/zen/…`) timing out with `OpenAI completions stream stalled while waiting for the next event` during slow plan-writing/reasoning phases. The 600s idle-timeout floor for GLM coding-plan SKUs was gated to the native Z.AI/Zhipu hosts only, so OpenCode-fronted GLM fell back to the 120s default watchdog. ([#4758](https://github.com/can1357/oh-my-pi/issues/4758)) + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index d5627ccd5..6f1dd0ed8 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -377,7 +377,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // for minutes while reasoning or cold-loading weights; widen the idle // timeout so warm-ups stop aborting and retrying. const streamIdleTimeoutMs = - GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu) + GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu || isOpenCodeHost) ? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS : provider === "alibaba-coding-plan" ? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index 82ff798be..b69b9afb7 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -120,6 +120,39 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { }); }); +describe("openai-completions compat — GLM coding-plan stream idle timeout", () => { + function glm52(provider: string, baseUrl: string): ModelSpec<"openai-completions"> { + return { ...baseModel, id: "glm-5.2", name: "GLM-5.2", provider, baseUrl }; + } + + // GLM coding-plan SKUs idle for minutes mid-reasoning; the 600s watchdog + // floor must apply on every gateway that fronts them, not just the native + // Z.AI/Zhipu hosts (issue #4758: GLM-5.2 via opencode-go stalled with + // "OpenAI completions stream stalled while waiting for the next event"). + it("widens the idle timeout to 600s for GLM-5.x on Z.AI, Zhipu, and OpenCode gateways", () => { + expect(buildOpenAICompat(glm52("zai", "https://api.z.ai/api/coding/paas/v4")).streamIdleTimeoutMs).toBe(600_000); + expect( + buildOpenAICompat(glm52("zhipu-coding-plan", "https://open.bigmodel.cn/api/coding/paas/v4")) + .streamIdleTimeoutMs, + ).toBe(600_000); + expect(buildOpenAICompat(glm52("opencode-go", "https://opencode.ai/zen/go/v1")).streamIdleTimeoutMs).toBe( + 600_000, + ); + expect(buildOpenAICompat(glm52("opencode-zen", "https://opencode.ai/zen/v1")).streamIdleTimeoutMs).toBe(600_000); + }); + + it("does not widen non-GLM models on the OpenCode gateway via the GLM floor", () => { + const kimi = buildOpenAICompat({ + ...baseModel, + id: "kimi-k2.5", + name: "Kimi K2.5", + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + }); + expect(kimi.streamIdleTimeoutMs).toBeUndefined(); + }); +}); + describe("zhipu-coding-plan model discovery", () => { it("uses the dedicated Coding Plan endpoint by default", async () => { let requestedUrl = "";