fix(catalog): advertise xhigh on first-party grok-4.6 Responses

xAI documents xhigh on grok-4.6. Keep 4.5/4.3/3-mini on the 4-tier
ladder and leave leftover xhigh unmapped for 4.6, matching multi-agent.
This commit is contained in:
Yang Yang
2026-08-14 22:41:26 -07:00
parent a7ac5d9fd3
commit d02aa3c85f
8 changed files with 94 additions and 21 deletions
+2 -2
View File
@@ -17,10 +17,10 @@
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok 4.5 / 4.3 / 3-mini Responses rows; leftover `xhigh`/`max` requests clamp to `high`. `grok-4.20-multi-agent*` still advertises unmapped `xhigh` (16-agent mode).
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok 4.5 / 4.3 / 3-mini Responses rows; leftover `xhigh`/`max` requests clamp to `high`. `grok-4.6*` and `grok-4.20-multi-agent*` advertise unmapped `xhigh`.
- Stopped baking `reasoningEffortMap` on first-party xAI catalog rows that omit `reasoning.effort` (`omitReasoningEffort: true`).
- Suppressed presence/frequency penalties on every first-party xAI Responses model, including non-reasoning ids such as `grok-2`; xAI's `/v1/responses` marks those fields unsupported.
- Routed `grok-4.6` (added on main) through first-party xAI Responses with the same 4-tier effort allowlist as `grok-4.5`, instead of leaving the paid row on Chat Completions.
- Routed `grok-4.6` (added on main) through first-party xAI Responses and advertised its documented `xhigh` effort tier (4.5 stays 4-tier).
## [17.3.4] - 2026-08-14
+6 -6
View File
@@ -15,8 +15,8 @@ import {
isClaudeModelId,
isDeepseekModelIdOrName,
isGlm52ReasoningEffortModelId,
isGrokMultiAgentModelId,
isGrokReasoningEffortCapable,
isGrokXHighEffortCapable,
isKimiK3ModelId,
isKimiK26ModelId,
isKimiModelId,
@@ -178,11 +178,11 @@ const MIMO_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]>
xhigh: "high",
};
/** Shared `minimal → low` clamp. Multi-agent Grok keeps `xhigh` unmapped. */
/** Shared `minimal → low` clamp. xhigh-capable Grok keeps `xhigh` unmapped. */
const XAI_RESPONSES_MINIMAL_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
minimal: "low",
};
/** Non-multi-agent Grok: leftover `xhigh`/`max` clamp to `high` (4.5 has no 16-agent mode). */
/** Grok 4.5 / 4.3 / 3-mini: leftover `xhigh`/`max` clamp to `high`. */
const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
minimal: "low",
xhigh: "high",
@@ -191,7 +191,7 @@ const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffor
/** Wire effort remap for first-party xAI Responses. */
export function xaiResponsesReasoningEffortMap(modelId: string): NonNullable<OpenAICompat["reasoningEffortMap"]> {
return isGrokMultiAgentModelId(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
return isGrokXHighEffortCapable(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
}
function mergeModelReasoningEffortMap(
@@ -793,8 +793,8 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
if (isXaiHost) {
const canonical = xaiResponsesReasoningEffortMap(id);
compat.reasoningEffortMap = { ...compat.reasoningEffortMap, ...canonical };
// Multi-agent Grok advertises unmapped `xhigh`; drop a stale clamp from
// previous snapshots so 16-agent mode is not rewritten to `high`.
// xhigh-capable Grok advertises unmapped `xhigh`; drop a stale clamp
// from previous snapshots so 4.6 / 16-agent mode is not rewritten to `high`.
for (const key of ["xhigh", "max"] as const) {
if (!(key in canonical)) {
delete compat.reasoningEffortMap[key];
+13 -1
View File
@@ -132,12 +132,24 @@ export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
/**
* `grok-4.20-multi-agent*` uses `reasoning.effort` to pick agent count
* (`xhigh` is the 16-agent mode). Other first-party Grok effort SKUs stay on
* `low|medium|high` (https://docs.x.ai/developers/model-capabilities/text/reasoning).
* `low|medium|high` unless {@link isGrokXHighEffortCapable} (currently
* `grok-4.6*` plus multi-agent).
* https://docs.x.ai/developers/model-capabilities/text/reasoning
*/
export const isGrokMultiAgentModelId = memo((modelId: string): boolean => {
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.20-multi-agent");
});
/**
* First-party Grok SKUs whose Responses wire accepts `reasoning.effort: "xhigh"`.
* `grok-4.6*` documents xhigh as a reasoning depth; multi-agent uses it as
* 16-agent mode. `grok-4.5` / `grok-4.3` / `grok-3-mini` do not.
*/
export const isGrokXHighEffortCapable = memo((modelId: string): boolean => {
if (isGrokMultiAgentModelId(modelId)) return true;
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.6");
});
/**
* MiniMax M2-generation family (M2, M2.1, M2.5, M2.7, including `-highspeed`/
* `-lightning`/`-her`/`-turbo` variants, dotless aliases like `minimax-m21`,
+4 -4
View File
@@ -26,7 +26,7 @@ import {
isDeepseekModelIdOrName,
isDeepseekV4FlashModelId,
isGlm52ReasoningEffortModelId,
isGrokMultiAgentModelId,
isGrokXHighEffortCapable,
isKimiK3ModelId,
isMimoModelIdOrName,
isMinimaxM2FamilyModelId,
@@ -396,10 +396,10 @@ function getModelDefinedEfforts<TApi extends Api>(
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
return HIGH_MAX_REASONING_EFFORTS;
}
// First-party Grok: `grok-4.20-multi-agent*` advertises `xhigh` (16-agent
// mode). Other effort-capable SKUs stay on `minimal/low/medium/high`.
// First-party Grok: `grok-4.6*` and `grok-4.20-multi-agent*` advertise
// `xhigh`. Other effort-capable SKUs stay on `minimal/low/medium/high`.
if (modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "xai")) {
return isGrokMultiAgentModelId(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
return isGrokXHighEffortCapable(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
}
return isOpenAICompatReasoningApi(spec.api) &&
(isMinimaxM2FamilyModelId(spec.id) ||
+6 -8
View File
@@ -104468,7 +104468,8 @@
"minimal",
"low",
"medium",
"high"
"high",
"xhigh"
],
"effortMap": {
"minimal": "low"
@@ -104476,9 +104477,7 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low",
"xhigh": "high",
"max": "high"
"minimal": "low"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104799,7 +104798,8 @@
"minimal",
"low",
"medium",
"high"
"high",
"xhigh"
],
"effortMap": {
"minimal": "low"
@@ -104809,9 +104809,7 @@
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low",
"xhigh": "high",
"max": "high"
"minimal": "low"
},
"includeEncryptedReasoning": true,
"filterReasoningHistory": false,
@@ -7,6 +7,7 @@ import {
isGrokModelId,
isGrokMultiAgentModelId,
isGrokReasoningEffortCapable,
isGrokXHighEffortCapable,
isKimiK26ModelId,
isKimiModelId,
isMinimaxM2FamilyModelId,
@@ -361,7 +362,25 @@ describe("isGrokMultiAgentModelId", () => {
test("rejects other Grok ids", () => {
expect(isGrokMultiAgentModelId("grok-4.5")).toBe(false);
expect(isGrokMultiAgentModelId("grok-4.6")).toBe(false);
expect(isGrokMultiAgentModelId("grok-4.20-0309-reasoning")).toBe(false);
expect(isGrokMultiAgentModelId("")).toBe(false);
});
});
describe("isGrokXHighEffortCapable", () => {
test("matches grok-4.6 and multi-agent SKUs across namespaces", () => {
expect(isGrokXHighEffortCapable("grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("xai/grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("xai-oauth/grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("grok-4.20-multi-agent-0309")).toBe(true);
});
test("rejects Grok SKUs that clamp leftover xhigh to high", () => {
expect(isGrokXHighEffortCapable("grok-4.5")).toBe(false);
expect(isGrokXHighEffortCapable("grok-4.3")).toBe(false);
expect(isGrokXHighEffortCapable("grok-3-mini")).toBe(false);
expect(isGrokXHighEffortCapable("grok-build")).toBe(false);
expect(isGrokXHighEffortCapable("")).toBe(false);
});
});
@@ -903,6 +903,26 @@ describe("model thinking runtime helpers", () => {
expect(() => requireSupportedEffort(paid, Effort.XHigh)).toThrow(/not supported/);
});
it("exposes xhigh on first-party Grok 4.6 Responses models", () => {
const paid = createModel({
id: "grok-4.6",
api: "openai-responses",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
});
const oauth = createModel({
id: "grok-4.6",
api: "openai-responses",
provider: "xai-oauth",
baseUrl: "https://api.x.ai/v1",
});
expect(paid.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
expect(oauth.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
expect(requireSupportedEffort(paid, Effort.XHigh)).toBe(Effort.XHigh);
expect(paid.compat.reasoningEffortMap?.xhigh).toBeUndefined();
});
it("exposes xhigh on first-party Grok multi-agent Responses models", () => {
const paid = createModel({
id: "grok-4.20-multi-agent-beta-latest",
@@ -19,6 +19,14 @@ const XAI_MODELS_DEV_FIXTURE = {
limit: { context: 500_000, output: 500_000 },
cost: { input: 2, output: 6, cache_read: 0.3 },
},
"grok-4.6": {
name: "Grok 4.6",
tool_call: true,
reasoning: true,
modalities: { input: ["text", "image"] },
limit: { context: 500_000, output: 500_000 },
cost: { input: 2, output: 6, cache_read: 0.5 },
},
"grok-code-fast-1": {
name: "Grok Code Fast 1",
tool_call: true,
@@ -109,6 +117,18 @@ describe("paid xAI Responses thinking policy", () => {
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
effortMap: { minimal: "low" },
});
expect(byId["grok-4.6"]?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
effortMap: { minimal: "low" },
});
expect(byId["grok-4.6"]?.compat).toMatchObject({
supportsReasoningEffort: true,
reasoningEffortMap: { minimal: "low" },
});
expect(byId["grok-4.6"]?.compat).not.toMatchObject({
reasoningEffortMap: { xhigh: "high" },
});
expect(byId["grok-4.20-multi-agent-beta-latest"]?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
@@ -140,6 +160,10 @@ describe("paid xAI Responses thinking policy", () => {
expect(bundled["grok-4.5"]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(bundled["grok-4.5"]?.thinking?.efforts).not.toContain(Effort.XHigh);
expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true);
expect(bundled["grok-4.6"]?.thinking?.efforts).toContain(Effort.XHigh);
expect(bundled["grok-4.6"]?.compat).not.toMatchObject({
reasoningEffortMap: { xhigh: "high" },
});
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.thinking?.efforts).toContain(Effort.XHigh);
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.compat).not.toMatchObject({
reasoningEffortMap: { xhigh: "high" },