fix(ai): omitted max token capping for OpenRouter completions
- Added OpenRouter detection in openai-completions parameter construction and skipped max-token emission for non-Kimi OpenRouter routes. - Preserved existing 64k-cap clamping behavior for non-OpenRouter requests and model-specific limits. - Expanded max-output-token tests to cover OpenRouter omission behavior and Kimi-over-OpenRouter token handling.
This commit is contained in:
@@ -9,7 +9,8 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed a degenerate OpenAI Codex stream (the model emits whitespace-only `function_call_arguments.delta` frames forever — commonly seen right after a `todo` tool call) terminating the turn with an error instead of recovering. The whitespace-loop circuit-breaker now (a) stops aborting the shared per-request `AbortController` — `requestSignal` is an `AbortSignal.any` over it, so aborting latched it and made every reopen on the reused `requestSetup` impossible — and (b) drops the half-built junk tool call and replays the request from scratch, bounded by `CODEX_WHITESPACE_LOOP_RETRY_LIMIT` (2). Sampling nondeterminism usually clears the loop on a fresh attempt; once the budget is exhausted the error is surfaced as before, but without the junk tool call polluting the message.
|
||||
- Capped requested output tokens at 64k (`OPENAI_MAX_OUTPUT_TOKENS`, mirroring Anthropic's `CLAUDE_CODE_MAX_OUTPUT_TOKENS`) across every OpenAI-family wire — the `openai-completions` request builder and the shared responses sampling helper (`openai-responses`, `azure-openai-responses`). A model's catalog `maxTokens` often reflects its context window rather than a given upstream's real per-request output cap: OpenRouter advertises 131072 output tokens for `z-ai/glm-4.7`, but the Cerebras upstream only allows ~131072 tokens *total*, so requesting the full ceiling as output 400'd with "maximum context length is 131072 tokens". Output is now clamped to `min(requested, model.maxTokens, 64000)`.
|
||||
- Capped requested output tokens at 64k (`OPENAI_MAX_OUTPUT_TOKENS`, mirroring Anthropic's `CLAUDE_CODE_MAX_OUTPUT_TOKENS`) on OpenAI-family wires with a known upstream output cap — the `openai-completions` request builder (non-OpenRouter) and the shared responses sampling helper (`openai-responses`, `azure-openai-responses`). A model's catalog `maxTokens` often tracks its context window rather than the upstream's per-request output cap, so requesting the full ceiling 400'd (e.g. `z-ai/glm-4.7` asking for 131072 output exceeded the upstream's 131072-token *total* context). Output is now `min(requested, model.maxTokens, 64000)`.
|
||||
- Stopped sending `max_tokens`/`max_completion_tokens` on OpenRouter (`openrouter.ai`) completions requests. OpenRouter filters out any upstream whose advertised output cap is below the requested `max_tokens`, so a value derived from the catalog (which reflects the highest-cap provider) silently excluded lower-cap upstreams — `provider.order: ["cerebras"]` for `z-ai/glm-4.7` fell through to DeepInfra because Cerebras's ~40k output cap is below the request, while `only: ["cerebras"]` (no fallback target) bypassed the filter and worked. Omitting the field lets each upstream self-cap and keeps provider routing (`only`/`order`) honored. Kimi via OpenRouter stays exempt — it derives TPM rate limits from `max_tokens`.
|
||||
|
||||
## [15.10.7] - 2026-06-08
|
||||
|
||||
|
||||
@@ -1204,6 +1204,7 @@ function buildParams(
|
||||
compat.reasoningContentField = "reasoning_content";
|
||||
}
|
||||
const isKimiModelId = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
|
||||
const isOpenRouter = model.baseUrl.includes("openrouter.ai");
|
||||
const messages = convertMessages(model, context, compat);
|
||||
maybeAddAnthropicCacheControl(compat, messages);
|
||||
const supportsReasoningParams = model.provider !== "github-copilot";
|
||||
@@ -1217,8 +1218,15 @@ function buildParams(
|
||||
// Kimi-family regex used by the compat detector.
|
||||
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
|
||||
const requestedMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined);
|
||||
// OpenRouter fans out to upstreams whose output caps differ from the catalog
|
||||
// value (which tracks the highest-cap provider). A max_tokens above the routed
|
||||
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
|
||||
// GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit
|
||||
// it for OpenRouter so each upstream self-caps and routing is honored. Kimi is
|
||||
// exempt — it derives TPM rate limits from max_tokens (see above).
|
||||
const omitMaxTokensForRouting = isOpenRouter && !isKimiModelId;
|
||||
const effectiveMaxTokens =
|
||||
requestedMaxTokens === undefined
|
||||
requestedMaxTokens === undefined || omitMaxTokensForRouting
|
||||
? undefined
|
||||
: Math.min(requestedMaxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS);
|
||||
|
||||
|
||||
@@ -2,16 +2,17 @@ import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-ai/models";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import { OPENAI_MAX_OUTPUT_TOKENS, type Context, type Model } from "@oh-my-pi/pi-ai/types";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
|
||||
import { type Context, type Model, OPENAI_MAX_OUTPUT_TOKENS } from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
// Regression for the OpenRouter -> Cerebras GLM-4.7 overflow: the catalog
|
||||
// `maxTokens` (131072) reflected the model's window, not the Cerebras upstream's
|
||||
// per-request limit, so omp requested the full ceiling as output and 400'd.
|
||||
// Output is now clamped to OPENAI_MAX_OUTPUT_TOKENS (mirroring Anthropic's cap)
|
||||
// across both OpenAI-family wires: responses (applyCommonResponsesSamplingParams)
|
||||
// and completions (streamOpenAICompletions).
|
||||
|
||||
const originalFetch = global.fetch;
|
||||
// Output-token wire policy for OpenAI-family providers:
|
||||
// - Non-aggregator completions + all responses: clamp to OPENAI_MAX_OUTPUT_TOKENS
|
||||
// (mirrors Anthropic's cap) so a catalog maxTokens that tracks the context
|
||||
// window never overflows the upstream.
|
||||
// - OpenRouter completions: omit max_tokens entirely. OpenRouter filters out any
|
||||
// upstream whose output cap is below the requested value (e.g. Cerebras GLM-4.7
|
||||
// ~40k), silently defeating provider routing. Kimi via OpenRouter is exempt
|
||||
// (it derives TPM rate limits from max_tokens).
|
||||
|
||||
const ctx: Context = {
|
||||
systemPrompt: ["hi"],
|
||||
@@ -19,13 +20,12 @@ const ctx: Context = {
|
||||
};
|
||||
|
||||
afterEach(() => {
|
||||
global.fetch = originalFetch;
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
function captureResponsesBody(): Record<string, unknown> {
|
||||
function captureResponsesBody(): { fetchMock: FetchImpl; captured: Record<string, unknown> } {
|
||||
const captured: Record<string, unknown> = {};
|
||||
const fetchMock = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
const fetchMock: FetchImpl = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
|
||||
Object.assign(captured, body);
|
||||
const event = {
|
||||
@@ -40,13 +40,12 @@ function captureResponsesBody(): Record<string, unknown> {
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
});
|
||||
global.fetch = Object.assign(fetchMock, { preconnect: originalFetch.preconnect }) as typeof fetch;
|
||||
return captured;
|
||||
return { fetchMock, captured };
|
||||
}
|
||||
|
||||
async function drainResponses(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
|
||||
const captured = captureResponsesBody();
|
||||
const stream = streamSimple(model, ctx, { apiKey: "k" });
|
||||
const { fetchMock, captured } = captureResponsesBody();
|
||||
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock });
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
@@ -55,8 +54,20 @@ async function drainResponses(model: Model<"openai-responses">): Promise<Record<
|
||||
|
||||
function completionsSse(): Response {
|
||||
const events: unknown[] = [
|
||||
{ id: "c", object: "chat.completion.chunk", created: 0, model: "m", choices: [{ index: 0, delta: { content: "ok" } }] },
|
||||
{ id: "c", object: "chat.completion.chunk", created: 0, model: "m", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] },
|
||||
{
|
||||
id: "c",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: "m",
|
||||
choices: [{ index: 0, delta: { content: "ok" } }],
|
||||
},
|
||||
{
|
||||
id: "c",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: "m",
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
},
|
||||
"[DONE]",
|
||||
];
|
||||
const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`;
|
||||
@@ -68,15 +79,12 @@ async function captureCompletionsBody(
|
||||
maxTokens: number,
|
||||
): Promise<Record<string, unknown>> {
|
||||
let payload: Record<string, unknown> | undefined;
|
||||
global.fetch = Object.assign(
|
||||
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
|
||||
return completionsSse();
|
||||
},
|
||||
{ preconnect: originalFetch.preconnect },
|
||||
) as typeof fetch;
|
||||
const fetchMock: FetchImpl = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
|
||||
return completionsSse();
|
||||
};
|
||||
|
||||
const result = await streamOpenAICompletions(model, ctx, { apiKey: "k", maxTokens }).result();
|
||||
const result = await streamOpenAICompletions(model, ctx, { apiKey: "k", maxTokens, fetch: fetchMock }).result();
|
||||
expect(result.stopReason).toBe("stop");
|
||||
if (!payload) throw new Error("Expected OpenAI completions request payload");
|
||||
return payload;
|
||||
@@ -98,6 +106,38 @@ function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> {
|
||||
};
|
||||
}
|
||||
|
||||
// Non-aggregator completions model: the 64k clamp applies (max_tokens is sent).
|
||||
function directCompletionsModel(maxTokens: number): Model<"openai-completions"> {
|
||||
return {
|
||||
id: "glm-4.7",
|
||||
name: "GLM 4.7 (direct)",
|
||||
api: "openai-completions",
|
||||
provider: "cerebras",
|
||||
baseUrl: "https://api.cerebras.ai/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131_072,
|
||||
maxTokens,
|
||||
};
|
||||
}
|
||||
|
||||
// Kimi via OpenRouter stays exempt from the omit (TPM rate limits need max_tokens).
|
||||
function kimiOpenRouterModel(maxTokens: number): Model<"openai-completions"> {
|
||||
return {
|
||||
id: "moonshotai/kimi-k2.5",
|
||||
name: "Kimi K2.5",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131_072,
|
||||
maxTokens,
|
||||
};
|
||||
}
|
||||
|
||||
describe("OpenAI-family output-token cap", () => {
|
||||
it("clamps openai-responses max_output_tokens to the 64k ceiling", async () => {
|
||||
const model: Model<"openai-responses"> = {
|
||||
@@ -109,18 +149,29 @@ describe("OpenAI-family output-token cap", () => {
|
||||
expect(body.max_output_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
|
||||
});
|
||||
|
||||
it("clamps openai-completions output tokens to the 64k ceiling (OpenRouter GLM-4.7 repro)", async () => {
|
||||
const body = await captureCompletionsBody(glmCompletionsModel(131_072), 131_072);
|
||||
it("clamps non-aggregator completions output to the 64k ceiling", async () => {
|
||||
const body = await captureCompletionsBody(directCompletionsModel(131_072), 131_072);
|
||||
expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
|
||||
});
|
||||
|
||||
it("never raises a requested output below the ceiling", async () => {
|
||||
const body = await captureCompletionsBody(glmCompletionsModel(131_072), 8_000);
|
||||
const body = await captureCompletionsBody(directCompletionsModel(131_072), 8_000);
|
||||
expect(body.max_completion_tokens ?? body.max_tokens).toBe(8_000);
|
||||
});
|
||||
|
||||
it("respects a model maxTokens that is below the ceiling", async () => {
|
||||
const body = await captureCompletionsBody(glmCompletionsModel(32_000), 131_072);
|
||||
const body = await captureCompletionsBody(directCompletionsModel(32_000), 131_072);
|
||||
expect(body.max_completion_tokens ?? body.max_tokens).toBe(32_000);
|
||||
});
|
||||
|
||||
it("omits max_tokens entirely for OpenRouter so provider routing is not filtered", async () => {
|
||||
const body = await captureCompletionsBody(glmCompletionsModel(131_072), 131_072);
|
||||
expect(body.max_tokens).toBeUndefined();
|
||||
expect(body.max_completion_tokens).toBeUndefined();
|
||||
});
|
||||
|
||||
it("still sends max_tokens for Kimi via OpenRouter (TPM rate-limit requirement)", async () => {
|
||||
const body = await captureCompletionsBody(kimiOpenRouterModel(131_072), 131_072);
|
||||
expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user