fix(catalog): omit Responses penalties on all first-party xAI models

xAI's /v1/responses rejects presence/frequency penalties for every Grok
model, not only reasoners. Gate supportsPenaltyAndStopParams on isXaiHost
so xai/grok-2 no longer serializes presence_penalty.
This commit is contained in:
Yang Yang
2026-08-09 22:55:10 -07:00
parent 76faddf886
commit 86866b8bbf
7 changed files with 25 additions and 9 deletions
+1
View File
@@ -242,6 +242,7 @@ Both the paid API-key provider (`xai` / `XAI_API_KEY`) and SuperGrok OAuth
- omit `reasoning.effort` unless the model is on the Grok effort-capable allowlist
- omit `reasoning.summary` (the host rejects it; do not fall back to `"auto"`)
- omit presence/frequency penalties (`/v1/responses` rejects them for every Grok model)
- include `reasoning.encrypted_content` on the request
- replay encrypted reasoning items on later turns
+1
View File
@@ -6,6 +6,7 @@
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
- Omitted presence/frequency penalties on all first-party xAI Responses models, including non-reasoning ids such as `xai/grok-2`.
## [17.3.4] - 2026-08-14
@@ -121,6 +121,16 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
expect(params.temperature).toBe(0.2);
});
test("paid xai/grok-2 omits presence_penalty on non-reasoning Responses models", () => {
const grok2 = getBundledModel<"openai-responses">("xai", "grok-2");
if (!grok2) throw new Error("xai/grok-2 must be in bundled models.json");
const { params } = buildParams(grok2, singleUserContext, { presencePenalty: 0.4, temperature: 0.2 }, undefined);
expect(params).not.toHaveProperty("presence_penalty");
expect(params.temperature).toBe(0.2);
});
test("paid xai/grok-4.5 clamps minimal reasoning effort to low", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
+1
View File
@@ -19,6 +19,7 @@
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok Responses catalog rows; leftover `xhigh`/`max` requests clamp to `high`.
- Stopped baking `reasoningEffortMap` on first-party xAI catalog rows that omit `reasoning.effort` (`omitReasoningEffort: true`).
- Suppressed presence/frequency penalties on every first-party xAI Responses model, including non-reasoning ids such as `grok-2`; xAI's `/v1/responses` marks those fields unsupported.
## [17.3.4] - 2026-08-14
+3 -3
View File
@@ -722,9 +722,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// temperature/top_p/… with a 400 on every serving host (#5606).
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id),
// xAI reasoning models 400 on presence/frequency penalties and stop
// (https://docs.x.ai/developers/model-capabilities/text/reasoning).
supportsPenaltyAndStopParams: !(isXaiHost && reasoningCapable),
// xAI `/v1/responses` rejects presence/frequency penalties for every
// model, not only reasoners (https://docs.x.ai/developers/rest-api-reference/inference/chat).
supportsPenaltyAndStopParams: !isXaiHost,
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
+3 -2
View File
@@ -365,8 +365,9 @@ export interface OpenAICompat {
supportsSamplingParams?: boolean;
/**
* Whether presence/frequency penalties and stop sequences may be sent.
* xAI reasoning models reject `presencePenalty`, `frequencyPenalty`, and
* `stop` with a 400. When unset, auto-detected. Default: true.
* First-party xAI `/v1/responses` rejects penalty fields for every model.
* xAI reasoning models also reject them (and `stop`) on chat completions.
* When unset, auto-detected. Default: true.
*/
supportsPenaltyAndStopParams?: boolean;
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
+6 -4
View File
@@ -272,12 +272,14 @@ describe("xAI Responses reasoning-effort suppression", () => {
expect(oauth.supportsReasoningSummary).toBe(false);
});
it("keeps penalty and stop params on non-reasoning paid xAI models", () => {
const compat = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-4-fast-non-reasoning", "xai"),
it("suppresses penalty params on every first-party xAI Responses model", () => {
const reasoning = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.5", "xai"));
const nonReasoning = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-2", "xai"),
reasoning: false,
});
expect(compat.supportsPenaltyAndStopParams).toBe(true);
expect(reasoning.supportsPenaltyAndStopParams).toBe(false);
expect(nonReasoning.supportsPenaltyAndStopParams).toBe(false);
});
it("omits effort for paid xai models off the Grok allowlist", () => {