From af33e4055ec4e73e3a84c1506c8b2217147818de Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 9 Jun 2026 04:04:50 +0200 Subject: [PATCH] fix(ai): omitted max token capping for OpenRouter completions - Added OpenRouter detection in openai-completions parameter construction and skipped max-token emission for non-Kimi OpenRouter routes. - Preserved existing 64k-cap clamping behavior for non-OpenRouter requests and model-specific limits. - Expanded max-output-token tests to cover OpenRouter omission behavior and Kimi-over-OpenRouter token handling. --- packages/ai/CHANGELOG.md | 3 +- .../ai/src/providers/openai-completions.ts | 10 +- .../test/openai-max-output-tokens-cap.test.ts | 111 +++++++++++++----- 3 files changed, 92 insertions(+), 32 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 0dc10d2ca..752aa7375 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -9,7 +9,8 @@ ### Fixed - Fixed a degenerate OpenAI Codex stream (the model emits whitespace-only `function_call_arguments.delta` frames forever — commonly seen right after a `todo` tool call) terminating the turn with an error instead of recovering. The whitespace-loop circuit-breaker now (a) stops aborting the shared per-request `AbortController` — `requestSignal` is an `AbortSignal.any` over it, so aborting latched it and made every reopen on the reused `requestSetup` impossible — and (b) drops the half-built junk tool call and replays the request from scratch, bounded by `CODEX_WHITESPACE_LOOP_RETRY_LIMIT` (2). Sampling nondeterminism usually clears the loop on a fresh attempt; once the budget is exhausted the error is surfaced as before, but without the junk tool call polluting the message. -- Capped requested output tokens at 64k (`OPENAI_MAX_OUTPUT_TOKENS`, mirroring Anthropic's `CLAUDE_CODE_MAX_OUTPUT_TOKENS`) across every OpenAI-family wire — the `openai-completions` request builder and the shared responses sampling helper (`openai-responses`, `azure-openai-responses`). A model's catalog `maxTokens` often reflects its context window rather than a given upstream's real per-request output cap: OpenRouter advertises 131072 output tokens for `z-ai/glm-4.7`, but the Cerebras upstream only allows ~131072 tokens *total*, so requesting the full ceiling as output 400'd with "maximum context length is 131072 tokens". Output is now clamped to `min(requested, model.maxTokens, 64000)`. +- Capped requested output tokens at 64k (`OPENAI_MAX_OUTPUT_TOKENS`, mirroring Anthropic's `CLAUDE_CODE_MAX_OUTPUT_TOKENS`) on OpenAI-family wires with a known upstream output cap — the `openai-completions` request builder (non-OpenRouter) and the shared responses sampling helper (`openai-responses`, `azure-openai-responses`). A model's catalog `maxTokens` often tracks its context window rather than the upstream's per-request output cap, so requesting the full ceiling 400'd (e.g. `z-ai/glm-4.7` asking for 131072 output exceeded the upstream's 131072-token *total* context). Output is now `min(requested, model.maxTokens, 64000)`. +- Stopped sending `max_tokens`/`max_completion_tokens` on OpenRouter (`openrouter.ai`) completions requests. OpenRouter filters out any upstream whose advertised output cap is below the requested `max_tokens`, so a value derived from the catalog (which reflects the highest-cap provider) silently excluded lower-cap upstreams — `provider.order: ["cerebras"]` for `z-ai/glm-4.7` fell through to DeepInfra because Cerebras's ~40k output cap is below the request, while `only: ["cerebras"]` (no fallback target) bypassed the filter and worked. Omitting the field lets each upstream self-cap and keeps provider routing (`only`/`order`) honored. Kimi via OpenRouter stays exempt — it derives TPM rate limits from `max_tokens`. ## [15.10.7] - 2026-06-08 diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 2aa170e6e..808077012 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1204,6 +1204,7 @@ function buildParams( compat.reasoningContentField = "reasoning_content"; } const isKimiModelId = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id); + const isOpenRouter = model.baseUrl.includes("openrouter.ai"); const messages = convertMessages(model, context, compat); maybeAddAnthropicCacheControl(compat, messages); const supportsReasoningParams = model.provider !== "github-copilot"; @@ -1217,8 +1218,15 @@ function buildParams( // Kimi-family regex used by the compat detector. // Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts. const requestedMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined); + // OpenRouter fans out to upstreams whose output caps differ from the catalog + // value (which tracks the highest-cap provider). A max_tokens above the routed + // upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras + // GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit + // it for OpenRouter so each upstream self-caps and routing is honored. Kimi is + // exempt — it derives TPM rate limits from max_tokens (see above). + const omitMaxTokensForRouting = isOpenRouter && !isKimiModelId; const effectiveMaxTokens = - requestedMaxTokens === undefined + requestedMaxTokens === undefined || omitMaxTokensForRouting ? undefined : Math.min(requestedMaxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS); diff --git a/packages/ai/test/openai-max-output-tokens-cap.test.ts b/packages/ai/test/openai-max-output-tokens-cap.test.ts index d5e97729f..33dbe2b32 100644 --- a/packages/ai/test/openai-max-output-tokens-cap.test.ts +++ b/packages/ai/test/openai-max-output-tokens-cap.test.ts @@ -2,16 +2,17 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { getBundledModel } from "@oh-my-pi/pi-ai/models"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; -import { OPENAI_MAX_OUTPUT_TOKENS, type Context, type Model } from "@oh-my-pi/pi-ai/types"; +import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; +import { type Context, type Model, OPENAI_MAX_OUTPUT_TOKENS } from "@oh-my-pi/pi-ai/types"; -// Regression for the OpenRouter -> Cerebras GLM-4.7 overflow: the catalog -// `maxTokens` (131072) reflected the model's window, not the Cerebras upstream's -// per-request limit, so omp requested the full ceiling as output and 400'd. -// Output is now clamped to OPENAI_MAX_OUTPUT_TOKENS (mirroring Anthropic's cap) -// across both OpenAI-family wires: responses (applyCommonResponsesSamplingParams) -// and completions (streamOpenAICompletions). - -const originalFetch = global.fetch; +// Output-token wire policy for OpenAI-family providers: +// - Non-aggregator completions + all responses: clamp to OPENAI_MAX_OUTPUT_TOKENS +// (mirrors Anthropic's cap) so a catalog maxTokens that tracks the context +// window never overflows the upstream. +// - OpenRouter completions: omit max_tokens entirely. OpenRouter filters out any +// upstream whose output cap is below the requested value (e.g. Cerebras GLM-4.7 +// ~40k), silently defeating provider routing. Kimi via OpenRouter is exempt +// (it derives TPM rate limits from max_tokens). const ctx: Context = { systemPrompt: ["hi"], @@ -19,13 +20,12 @@ const ctx: Context = { }; afterEach(() => { - global.fetch = originalFetch; vi.restoreAllMocks(); }); -function captureResponsesBody(): Record { +function captureResponsesBody(): { fetchMock: FetchImpl; captured: Record } { const captured: Record = {}; - const fetchMock = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => { + const fetchMock: FetchImpl = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => { const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record) : {}; Object.assign(captured, body); const event = { @@ -40,13 +40,12 @@ function captureResponsesBody(): Record { headers: { "content-type": "text/event-stream" }, }); }); - global.fetch = Object.assign(fetchMock, { preconnect: originalFetch.preconnect }) as typeof fetch; - return captured; + return { fetchMock, captured }; } async function drainResponses(model: Model<"openai-responses">): Promise> { - const captured = captureResponsesBody(); - const stream = streamSimple(model, ctx, { apiKey: "k" }); + const { fetchMock, captured } = captureResponsesBody(); + const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock }); for await (const event of stream) { if (event.type === "done" || event.type === "error") break; } @@ -55,8 +54,20 @@ async function drainResponses(model: Model<"openai-responses">): Promise `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`; @@ -68,15 +79,12 @@ async function captureCompletionsBody( maxTokens: number, ): Promise> { let payload: Record | undefined; - global.fetch = Object.assign( - async (_input: string | URL | Request, init?: RequestInit): Promise => { - payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; - return completionsSse(); - }, - { preconnect: originalFetch.preconnect }, - ) as typeof fetch; + const fetchMock: FetchImpl = async (_input: string | URL | Request, init?: RequestInit): Promise => { + payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return completionsSse(); + }; - const result = await streamOpenAICompletions(model, ctx, { apiKey: "k", maxTokens }).result(); + const result = await streamOpenAICompletions(model, ctx, { apiKey: "k", maxTokens, fetch: fetchMock }).result(); expect(result.stopReason).toBe("stop"); if (!payload) throw new Error("Expected OpenAI completions request payload"); return payload; @@ -98,6 +106,38 @@ function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> { }; } +// Non-aggregator completions model: the 64k clamp applies (max_tokens is sent). +function directCompletionsModel(maxTokens: number): Model<"openai-completions"> { + return { + id: "glm-4.7", + name: "GLM 4.7 (direct)", + api: "openai-completions", + provider: "cerebras", + baseUrl: "https://api.cerebras.ai/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131_072, + maxTokens, + }; +} + +// Kimi via OpenRouter stays exempt from the omit (TPM rate limits need max_tokens). +function kimiOpenRouterModel(maxTokens: number): Model<"openai-completions"> { + return { + id: "moonshotai/kimi-k2.5", + name: "Kimi K2.5", + api: "openai-completions", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131_072, + maxTokens, + }; +} + describe("OpenAI-family output-token cap", () => { it("clamps openai-responses max_output_tokens to the 64k ceiling", async () => { const model: Model<"openai-responses"> = { @@ -109,18 +149,29 @@ describe("OpenAI-family output-token cap", () => { expect(body.max_output_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS); }); - it("clamps openai-completions output tokens to the 64k ceiling (OpenRouter GLM-4.7 repro)", async () => { - const body = await captureCompletionsBody(glmCompletionsModel(131_072), 131_072); + it("clamps non-aggregator completions output to the 64k ceiling", async () => { + const body = await captureCompletionsBody(directCompletionsModel(131_072), 131_072); expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS); }); it("never raises a requested output below the ceiling", async () => { - const body = await captureCompletionsBody(glmCompletionsModel(131_072), 8_000); + const body = await captureCompletionsBody(directCompletionsModel(131_072), 8_000); expect(body.max_completion_tokens ?? body.max_tokens).toBe(8_000); }); it("respects a model maxTokens that is below the ceiling", async () => { - const body = await captureCompletionsBody(glmCompletionsModel(32_000), 131_072); + const body = await captureCompletionsBody(directCompletionsModel(32_000), 131_072); expect(body.max_completion_tokens ?? body.max_tokens).toBe(32_000); }); + + it("omits max_tokens entirely for OpenRouter so provider routing is not filtered", async () => { + const body = await captureCompletionsBody(glmCompletionsModel(131_072), 131_072); + expect(body.max_tokens).toBeUndefined(); + expect(body.max_completion_tokens).toBeUndefined(); + }); + + it("still sends max_tokens for Kimi via OpenRouter (TPM rate-limit requirement)", async () => { + const body = await captureCompletionsBody(kimiOpenRouterModel(131_072), 131_072); + expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS); + }); });