fix(catalog): omit Responses penalties on all first-party xAI models
xAI's /v1/responses rejects presence/frequency penalties for every Grok model, not only reasoners. Gate supportsPenaltyAndStopParams on isXaiHost so xai/grok-2 no longer serializes presence_penalty.
This commit is contained in:
@@ -242,6 +242,7 @@ Both the paid API-key provider (`xai` / `XAI_API_KEY`) and SuperGrok OAuth
|
||||
|
||||
- omit `reasoning.effort` unless the model is on the Grok effort-capable allowlist
|
||||
- omit `reasoning.summary` (the host rejects it; do not fall back to `"auto"`)
|
||||
- omit presence/frequency penalties (`/v1/responses` rejects them for every Grok model)
|
||||
- include `reasoning.encrypted_content` on the request
|
||||
- replay encrypted reasoning items on later turns
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
|
||||
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
|
||||
- Omitted presence/frequency penalties on all first-party xAI Responses models, including non-reasoning ids such as `xai/grok-2`.
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
|
||||
@@ -121,6 +121,16 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
|
||||
expect(params.temperature).toBe(0.2);
|
||||
});
|
||||
|
||||
test("paid xai/grok-2 omits presence_penalty on non-reasoning Responses models", () => {
|
||||
const grok2 = getBundledModel<"openai-responses">("xai", "grok-2");
|
||||
if (!grok2) throw new Error("xai/grok-2 must be in bundled models.json");
|
||||
|
||||
const { params } = buildParams(grok2, singleUserContext, { presencePenalty: 0.4, temperature: 0.2 }, undefined);
|
||||
|
||||
expect(params).not.toHaveProperty("presence_penalty");
|
||||
expect(params.temperature).toBe(0.2);
|
||||
});
|
||||
|
||||
test("paid xai/grok-4.5 clamps minimal reasoning effort to low", () => {
|
||||
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
|
||||
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
|
||||
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok Responses catalog rows; leftover `xhigh`/`max` requests clamp to `high`.
|
||||
- Stopped baking `reasoningEffortMap` on first-party xAI catalog rows that omit `reasoning.effort` (`omitReasoningEffort: true`).
|
||||
- Suppressed presence/frequency penalties on every first-party xAI Responses model, including non-reasoning ids such as `grok-2`; xAI's `/v1/responses` marks those fields unsupported.
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
|
||||
@@ -722,9 +722,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
|
||||
// temperature/top_p/… with a 400 on every serving host (#5606).
|
||||
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id),
|
||||
// xAI reasoning models 400 on presence/frequency penalties and stop
|
||||
// (https://docs.x.ai/developers/model-capabilities/text/reasoning).
|
||||
supportsPenaltyAndStopParams: !(isXaiHost && reasoningCapable),
|
||||
// xAI `/v1/responses` rejects presence/frequency penalties for every
|
||||
// model, not only reasoners (https://docs.x.ai/developers/rest-api-reference/inference/chat).
|
||||
supportsPenaltyAndStopParams: !isXaiHost,
|
||||
thinkingFormat,
|
||||
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
|
||||
omitReasoningEffort: false,
|
||||
|
||||
@@ -365,8 +365,9 @@ export interface OpenAICompat {
|
||||
supportsSamplingParams?: boolean;
|
||||
/**
|
||||
* Whether presence/frequency penalties and stop sequences may be sent.
|
||||
* xAI reasoning models reject `presencePenalty`, `frequencyPenalty`, and
|
||||
* `stop` with a 400. When unset, auto-detected. Default: true.
|
||||
* First-party xAI `/v1/responses` rejects penalty fields for every model.
|
||||
* xAI reasoning models also reject them (and `stop`) on chat completions.
|
||||
* When unset, auto-detected. Default: true.
|
||||
*/
|
||||
supportsPenaltyAndStopParams?: boolean;
|
||||
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
|
||||
|
||||
@@ -272,12 +272,14 @@ describe("xAI Responses reasoning-effort suppression", () => {
|
||||
expect(oauth.supportsReasoningSummary).toBe(false);
|
||||
});
|
||||
|
||||
it("keeps penalty and stop params on non-reasoning paid xAI models", () => {
|
||||
const compat = buildOpenAIResponsesCompat({
|
||||
...grokResponsesSpec("grok-4-fast-non-reasoning", "xai"),
|
||||
it("suppresses penalty params on every first-party xAI Responses model", () => {
|
||||
const reasoning = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.5", "xai"));
|
||||
const nonReasoning = buildOpenAIResponsesCompat({
|
||||
...grokResponsesSpec("grok-2", "xai"),
|
||||
reasoning: false,
|
||||
});
|
||||
expect(compat.supportsPenaltyAndStopParams).toBe(true);
|
||||
expect(reasoning.supportsPenaltyAndStopParams).toBe(false);
|
||||
expect(nonReasoning.supportsPenaltyAndStopParams).toBe(false);
|
||||
});
|
||||
|
||||
it("omits effort for paid xai models off the Grok allowlist", () => {
|
||||
|
||||
Reference in New Issue
Block a user