Merge remote-tracking branch 'origin/farm/48457ba2/fix-zai-glm-plan-stream-stall'
This commit is contained in:
@@ -2,6 +2,11 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed GLM-5.x coding-plan OpenAI-compatible streams to use a longer default watchdog window, avoiding spurious `OpenAI completions stream stalled while waiting for the next event` errors during slow `glm-5.1` thinking/output phases. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494))
|
||||
- Fixed `zhipu-coding-plan` model discovery and credential validation to use the dedicated GLM Coding Plan endpoint (`https://open.bigmodel.cn/api/coding/paas/v4`) instead of the general BigModel endpoint, preventing requests from consuming ordinary account balance. ([#1494](https://github.com/can1357/oh-my-pi/issues/1494))
|
||||
|
||||
## [15.5.11] - 2026-05-29
|
||||
|
||||
### Added
|
||||
|
||||
@@ -870,7 +870,7 @@ export function zhipuCodingPlanModelManagerOptions(
|
||||
config?: ZhipuCodingPlanModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
const apiKey = config?.apiKey;
|
||||
const baseUrl = config?.baseUrl ?? "https://open.bigmodel.cn/api/paas/v4";
|
||||
const baseUrl = config?.baseUrl ?? "https://open.bigmodel.cn/api/coding/paas/v4";
|
||||
return {
|
||||
providerId: "zhipu-coding-plan",
|
||||
...(apiKey && {
|
||||
@@ -2676,13 +2676,18 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe
|
||||
},
|
||||
),
|
||||
// --- Zhipu Coding Plan ---
|
||||
openAiCompletionsDescriptor("zhipu-coding-plan", "zhipu-coding-plan", "https://open.bigmodel.cn/api/paas/v4", {
|
||||
compat: {
|
||||
thinkingFormat: "zai",
|
||||
reasoningContentField: "reasoning_content",
|
||||
supportsDeveloperRole: false,
|
||||
openAiCompletionsDescriptor(
|
||||
"zhipu-coding-plan",
|
||||
"zhipu-coding-plan",
|
||||
"https://open.bigmodel.cn/api/coding/paas/v4",
|
||||
{
|
||||
compat: {
|
||||
thinkingFormat: "zai",
|
||||
reasoningContentField: "reasoning_content",
|
||||
supportsDeveloperRole: false,
|
||||
},
|
||||
},
|
||||
}),
|
||||
),
|
||||
];
|
||||
|
||||
const filterActiveToolCallModels = (_id: string, m: ModelsDevModel): boolean => {
|
||||
|
||||
@@ -367,6 +367,25 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
||||
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
"OpenAI completions stream timed out while waiting for the first event";
|
||||
|
||||
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
|
||||
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
|
||||
|
||||
/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */
|
||||
export function getOpenAICompletionsStreamIdleTimeoutFallbackMs(
|
||||
model: Model<"openai-completions">,
|
||||
): number | undefined {
|
||||
if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined;
|
||||
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
model: Model<"openai-completions">,
|
||||
context: Context,
|
||||
@@ -387,7 +406,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
|
||||
try {
|
||||
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
||||
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
||||
const idleTimeoutMs =
|
||||
options?.streamIdleTimeoutMs ??
|
||||
getOpenAIStreamIdleTimeoutMs(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model));
|
||||
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
||||
const requestTimeoutMs =
|
||||
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
|
||||
|
||||
@@ -28,13 +28,14 @@ export function getStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_
|
||||
/**
|
||||
* Returns the idle timeout used for OpenAI-family streaming transports.
|
||||
*
|
||||
* `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` takes precedence over the generic
|
||||
* `PI_STREAM_IDLE_TIMEOUT_MS` because some deployments tune OpenAI-compatible
|
||||
* backends separately from Anthropic/Gemini-style transports.
|
||||
*
|
||||
* Set `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0` to disable the watchdog.
|
||||
*/
|
||||
export function getOpenAIStreamIdleTimeoutMs(): number | undefined {
|
||||
return normalizeIdleTimeoutMs(
|
||||
$env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_STREAM_IDLE_TIMEOUT_MS,
|
||||
DEFAULT_STREAM_IDLE_TIMEOUT_MS,
|
||||
);
|
||||
export function getOpenAIStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_TIMEOUT_MS): number | undefined {
|
||||
return normalizeIdleTimeoutMs($env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_STREAM_IDLE_TIMEOUT_MS, fallbackMs);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,19 +1,19 @@
|
||||
/**
|
||||
* Zhipu Coding Plan login flow.
|
||||
*
|
||||
* Zhipu BigModel (智谱) provides an OpenAI-compatible API.
|
||||
* API docs: https://docs.bigmodel.cn/cn/guide/develop/openai/introduction
|
||||
* GLM Coding Plan provides an OpenAI-compatible API on the dedicated coding
|
||||
* endpoint. API docs: https://docs.bigmodel.cn/cn/coding-plan/quick-start
|
||||
*
|
||||
* Simple API key flow:
|
||||
* 1. User gets their API key from https://open.bigmodel.cn
|
||||
* 1. User gets a Coding Plan API key from https://bigmodel.cn/coding-plan/personal/overview
|
||||
* 2. User pastes the API key into the CLI
|
||||
*/
|
||||
|
||||
import { validateOpenAICompatibleApiKey } from "./api-key-validation";
|
||||
import type { OAuthController } from "./types";
|
||||
|
||||
const AUTH_URL = "https://open.bigmodel.cn/usercenter/apikeys";
|
||||
const API_BASE_URL = "https://open.bigmodel.cn/api/paas/v4";
|
||||
const AUTH_URL = "https://bigmodel.cn/coding-plan/personal/overview";
|
||||
const API_BASE_URL = "https://open.bigmodel.cn/api/coding/paas/v4";
|
||||
const VALIDATION_MODEL = "glm-5.1";
|
||||
|
||||
/**
|
||||
@@ -30,7 +30,7 @@ export async function loginZhipuCodingPlan(options: OAuthController): Promise<st
|
||||
// Open browser to API keys page
|
||||
options.onAuth?.({
|
||||
url: AUTH_URL,
|
||||
instructions: "Copy your API key from the dashboard",
|
||||
instructions: "Copy your API key from the Coding Plan dashboard",
|
||||
});
|
||||
|
||||
// Prompt user to paste their API key
|
||||
|
||||
@@ -1,6 +1,10 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { getBundledModel } from "../src/models";
|
||||
import { isOpenAICompletionsProgressChunk, streamOpenAICompletions } from "../src/providers/openai-completions";
|
||||
import {
|
||||
getOpenAICompletionsStreamIdleTimeoutFallbackMs,
|
||||
isOpenAICompletionsProgressChunk,
|
||||
streamOpenAICompletions,
|
||||
} from "../src/providers/openai-completions";
|
||||
import type { Context, Model } from "../src/types";
|
||||
|
||||
const originalFetch = global.fetch;
|
||||
@@ -79,6 +83,36 @@ function createKeepaliveOnlyCompletionsResponse(modelId: string, signal: AbortSi
|
||||
afterEach(() => {
|
||||
global.fetch = originalFetch;
|
||||
});
|
||||
describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
it("widens GLM 5.1 coding-plan stream watchdogs", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "glm-5.1",
|
||||
name: "GLM-5.1",
|
||||
provider: "zhipu-coding-plan",
|
||||
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
|
||||
});
|
||||
|
||||
it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "glm-5.1",
|
||||
name: "GLM-5.1",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.z.ai/api/coding/paas/v4",
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
|
||||
});
|
||||
|
||||
it("keeps ordinary OpenAI-compatible models on the global timeout", () => {
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* Contract: `isOpenAICompletionsProgressChunk` decides whether a streamed chunk
|
||||
* resets the idle-watchdog deadline in `iterateWithIdleTimeout`. A false
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import {
|
||||
getOpenAIStreamIdleTimeoutMs,
|
||||
getStreamFirstEventTimeoutMs,
|
||||
getStreamIdleTimeoutMs,
|
||||
iterateWithIdleTimeout,
|
||||
@@ -56,6 +57,23 @@ describe("getStreamIdleTimeoutMs(fallbackMs)", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("getOpenAIStreamIdleTimeoutMs(fallbackMs)", () => {
|
||||
it("returns the per-provider fallback when OpenAI env vars are unset", () => {
|
||||
expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBe(600_000);
|
||||
});
|
||||
|
||||
it("lets PI_OPENAI_STREAM_IDLE_TIMEOUT_MS override the fallback before the generic env var", () => {
|
||||
Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "42";
|
||||
Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84";
|
||||
expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBe(84);
|
||||
});
|
||||
|
||||
it("treats PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0 as a watchdog disable", () => {
|
||||
Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "0";
|
||||
expect(getOpenAIStreamIdleTimeoutMs(600_000)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => {
|
||||
it("returns the per-provider fallback when env unset and idle timeout is undefined", () => {
|
||||
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(300_000);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { zhipuCodingPlanModelManagerOptions } from "@oh-my-pi/pi-ai/provider-models/openai-compat";
|
||||
import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-ai/providers/openai-completions-compat";
|
||||
import type { Model } from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
@@ -21,6 +22,11 @@ const baseModel: Omit<Model<"openai-completions">, "provider" | "baseUrl"> = {
|
||||
reasoning: true,
|
||||
};
|
||||
|
||||
const originalFetch = global.fetch;
|
||||
|
||||
afterEach(() => {
|
||||
global.fetch = originalFetch;
|
||||
});
|
||||
function zhipuByProvider(): Model<"openai-completions"> {
|
||||
return {
|
||||
...baseModel,
|
||||
@@ -78,3 +84,24 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => {
|
||||
expect(resolved.reasoningContentField).toBe("reasoning_content");
|
||||
});
|
||||
});
|
||||
|
||||
describe("zhipu-coding-plan model discovery", () => {
|
||||
it("uses the dedicated Coding Plan endpoint by default", async () => {
|
||||
let requestedUrl = "";
|
||||
const mockFetch = async (input: string | Request | URL): Promise<Response> => {
|
||||
requestedUrl = input instanceof Request ? input.url : String(input);
|
||||
return new Response(JSON.stringify({ data: [{ id: "glm-5.1", name: "GLM-5.1" }] }), {
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
};
|
||||
global.fetch = Object.assign(mockFetch, { preconnect: originalFetch.preconnect });
|
||||
|
||||
const options = zhipuCodingPlanModelManagerOptions({ apiKey: "test-key" });
|
||||
expect(typeof options.fetchDynamicModels).toBe("function");
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(requestedUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4/models");
|
||||
expect(models?.[0]?.id).toBe("glm-5.1");
|
||||
expect(models?.[0]?.baseUrl).toBe("https://open.bigmodel.cn/api/coding/paas/v4");
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user