fix(catalog): omit unsupported reasoning.summary on paid xAI Responses

First-party xAI /v1/responses rejects reasoning.summary. Bake
supportsReasoningSummary=false for both xai and xai-oauth so paid
grok-4.5 effort requests send only reasoning.effort, matching SuperGrok.
This commit is contained in:
Yang Yang
2026-08-02 22:58:13 -07:00
parent b49b5b88d2
commit 01db5b04ee
8 changed files with 51 additions and 9 deletions
+2 -1
View File
@@ -240,7 +240,8 @@ Reasoning fields are not interchangeable.
Both the paid API-key provider (`xai` / `XAI_API_KEY`) and SuperGrok OAuth
(`xai-oauth`) chat over `https://api.x.ai/v1/responses`. Keep these independent:
- omit `reasoning.effort`
- omit `reasoning.effort` unless the model is on the Grok effort-capable allowlist
- omit `reasoning.summary` (the host rejects it; do not fall back to `"auto"`)
- include `reasoning.encrypted_content` on the request
- replay encrypted reasoning items on later turns
+4
View File
@@ -152,6 +152,10 @@
### Fixed
- Fixed an issue where Ollama requests without a user-role message would fail to generate output or silently fail with a misleading error.
### Fixed
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
## [17.2.5] - 2026-08-03
@@ -1285,12 +1285,11 @@ export function buildParams(
filterReasoningHistory: options?.filterReasoningHistory,
omitReasoningEffort: options?.omitReasoningEffort,
});
const reasoningSummary =
model.provider === "xai-oauth"
? options?.reasoning === undefined
? undefined
: null
: options?.reasoningSummary;
const reasoningSummary = model.compat.supportsReasoningSummary
? options?.reasoningSummary
: options?.reasoning === undefined
? undefined
: null;
applyResponsesCompatPolicy(params, reasoningPolicy, {
reasoningSummary,
forceReasoningOff: options?.forceReasoningOff,
@@ -87,6 +87,16 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 omits unsupported reasoning summary", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.reasoning).toEqual({ effort: "high" });
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 requests encrypted reasoning content", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
@@ -117,7 +127,7 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.Minimal }, undefined);
expect(params.reasoning).toMatchObject({ effort: "low" });
expect(params.reasoning).toEqual({ effort: "low" });
});
test("xai-oauth/grok-4.5 clamps minimal reasoning effort to low", () => {
@@ -126,7 +136,7 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.Minimal }, undefined);
expect(params.reasoning).toMatchObject({ effort: "low" });
expect(params.reasoning).toEqual({ effort: "low" });
});
test("xai-oauth/grok-4.5 replays encrypted reasoning on the next turn", () => {
+17
View File
@@ -142,6 +142,23 @@
- Fixed dynamic discovery for the `deepseek-v4` model family (such as `deepseek-v4-flash-0731`) under `alibaba-token-plan` missing reasoning configuration and maximum thinking effort.
- Fixed GitHub Copilot dynamic discovery retaining stale bundled prices for default-context models instead of using the provider's reported default-tier prices.
## [17.2.5] - 2026-08-03
### Changed
- Switched the paid xAI provider (`xai` / `XAI_API_KEY`) from Chat Completions to the OpenAI Responses API (`POST https://api.x.ai/v1/responses`), matching SuperGrok `xai-oauth`. Prompt-cache affinity (`x-grok-conv-id`), reasoning-effort allowlisting, and encrypted-reasoning replay rules are now shared across both first-party xAI hosts.
- Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`.
- Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`.
- Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter.
- Replayed xAI encrypted reasoning items on later Responses turns instead of stripping `type: "reasoning"` history.
### Fixed
- Invalidated stale paid-xAI model-cache rows written under Chat Completions so the Responses migration takes effect immediately instead of waiting for TTL expiry.
- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400.
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
## [17.2.5] - 2026-08-03
### Fixed
+3
View File
@@ -713,6 +713,8 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
// Copilot host under a different provider id still clamps.
supportsImageDetailOriginal:
!isXaiHost && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
// api.x.ai rejects `reasoning.summary` (SuperGrok and paid key alike).
supportsReasoningSummary: !isXaiHost,
reasoningEffortMap: isXaiHost ? { ...XAI_RESPONSES_REASONING_EFFORT_MAP } : {},
supportsReasoningParams: true,
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
@@ -796,6 +798,7 @@ function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnly
strictResponsesPairing: compat.strictResponsesPairing,
supportsImageDetailOriginal: compat.supportsImageDetailOriginal,
supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
supportsReasoningSummary: compat.supportsReasoningSummary,
isVercelGatewayHost: compat.isVercelGatewayHost,
} satisfies ResponsesOnlyCompat;
}
+6
View File
@@ -719,6 +719,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
strictResponsesPairing: boolean;
supportsImageDetailOriginal: boolean;
supportsObfuscationOptOut: boolean;
/**
* Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
* rejects the field; handlers pass `null` so the wire omits it instead of
* filling `"auto"`.
*/
supportsReasoningSummary: boolean;
streamIdleTimeoutMs?: number;
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
+2
View File
@@ -268,6 +268,8 @@ describe("xAI Responses reasoning-effort suppression", () => {
expect(oauth.reasoningEffortMap).toEqual({ minimal: "low" });
expect(paid.supportsPenaltyAndStopParams).toBe(false);
expect(oauth.supportsPenaltyAndStopParams).toBe(false);
expect(paid.supportsReasoningSummary).toBe(false);
expect(oauth.supportsReasoningSummary).toBe(false);
});
it("keeps penalty and stop params on non-reasoning paid xAI models", () => {