fix(catalog): advertise xhigh on first-party grok-4.6 Responses
xAI documents xhigh on grok-4.6. Keep 4.5/4.3/3-mini on the 4-tier ladder and leave leftover xhigh unmapped for 4.6, matching multi-agent.
This commit is contained in:
@@ -17,10 +17,10 @@
|
||||
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
|
||||
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
|
||||
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
|
||||
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok 4.5 / 4.3 / 3-mini Responses rows; leftover `xhigh`/`max` requests clamp to `high`. `grok-4.20-multi-agent*` still advertises unmapped `xhigh` (16-agent mode).
|
||||
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok 4.5 / 4.3 / 3-mini Responses rows; leftover `xhigh`/`max` requests clamp to `high`. `grok-4.6*` and `grok-4.20-multi-agent*` advertise unmapped `xhigh`.
|
||||
- Stopped baking `reasoningEffortMap` on first-party xAI catalog rows that omit `reasoning.effort` (`omitReasoningEffort: true`).
|
||||
- Suppressed presence/frequency penalties on every first-party xAI Responses model, including non-reasoning ids such as `grok-2`; xAI's `/v1/responses` marks those fields unsupported.
|
||||
- Routed `grok-4.6` (added on main) through first-party xAI Responses with the same 4-tier effort allowlist as `grok-4.5`, instead of leaving the paid row on Chat Completions.
|
||||
- Routed `grok-4.6` (added on main) through first-party xAI Responses and advertised its documented `xhigh` effort tier (4.5 stays 4-tier).
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
|
||||
@@ -15,8 +15,8 @@ import {
|
||||
isClaudeModelId,
|
||||
isDeepseekModelIdOrName,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isGrokXHighEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
@@ -178,11 +178,11 @@ const MIMO_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]>
|
||||
xhigh: "high",
|
||||
};
|
||||
|
||||
/** Shared `minimal → low` clamp. Multi-agent Grok keeps `xhigh` unmapped. */
|
||||
/** Shared `minimal → low` clamp. xhigh-capable Grok keeps `xhigh` unmapped. */
|
||||
const XAI_RESPONSES_MINIMAL_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
|
||||
minimal: "low",
|
||||
};
|
||||
/** Non-multi-agent Grok: leftover `xhigh`/`max` clamp to `high` (4.5 has no 16-agent mode). */
|
||||
/** Grok 4.5 / 4.3 / 3-mini: leftover `xhigh`/`max` clamp to `high`. */
|
||||
const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
|
||||
minimal: "low",
|
||||
xhigh: "high",
|
||||
@@ -191,7 +191,7 @@ const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffor
|
||||
|
||||
/** Wire effort remap for first-party xAI Responses. */
|
||||
export function xaiResponsesReasoningEffortMap(modelId: string): NonNullable<OpenAICompat["reasoningEffortMap"]> {
|
||||
return isGrokMultiAgentModelId(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
|
||||
return isGrokXHighEffortCapable(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
|
||||
}
|
||||
|
||||
function mergeModelReasoningEffortMap(
|
||||
@@ -793,8 +793,8 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
if (isXaiHost) {
|
||||
const canonical = xaiResponsesReasoningEffortMap(id);
|
||||
compat.reasoningEffortMap = { ...compat.reasoningEffortMap, ...canonical };
|
||||
// Multi-agent Grok advertises unmapped `xhigh`; drop a stale clamp from
|
||||
// previous snapshots so 16-agent mode is not rewritten to `high`.
|
||||
// xhigh-capable Grok advertises unmapped `xhigh`; drop a stale clamp
|
||||
// from previous snapshots so 4.6 / 16-agent mode is not rewritten to `high`.
|
||||
for (const key of ["xhigh", "max"] as const) {
|
||||
if (!(key in canonical)) {
|
||||
delete compat.reasoningEffortMap[key];
|
||||
|
||||
@@ -132,12 +132,24 @@ export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
|
||||
/**
|
||||
* `grok-4.20-multi-agent*` uses `reasoning.effort` to pick agent count
|
||||
* (`xhigh` is the 16-agent mode). Other first-party Grok effort SKUs stay on
|
||||
* `low|medium|high` (https://docs.x.ai/developers/model-capabilities/text/reasoning).
|
||||
* `low|medium|high` unless {@link isGrokXHighEffortCapable} (currently
|
||||
* `grok-4.6*` plus multi-agent).
|
||||
* https://docs.x.ai/developers/model-capabilities/text/reasoning
|
||||
*/
|
||||
export const isGrokMultiAgentModelId = memo((modelId: string): boolean => {
|
||||
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.20-multi-agent");
|
||||
});
|
||||
|
||||
/**
|
||||
* First-party Grok SKUs whose Responses wire accepts `reasoning.effort: "xhigh"`.
|
||||
* `grok-4.6*` documents xhigh as a reasoning depth; multi-agent uses it as
|
||||
* 16-agent mode. `grok-4.5` / `grok-4.3` / `grok-3-mini` do not.
|
||||
*/
|
||||
export const isGrokXHighEffortCapable = memo((modelId: string): boolean => {
|
||||
if (isGrokMultiAgentModelId(modelId)) return true;
|
||||
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.6");
|
||||
});
|
||||
|
||||
/**
|
||||
* MiniMax M2-generation family (M2, M2.1, M2.5, M2.7, including `-highspeed`/
|
||||
* `-lightning`/`-her`/`-turbo` variants, dotless aliases like `minimax-m21`,
|
||||
|
||||
@@ -26,7 +26,7 @@ import {
|
||||
isDeepseekModelIdOrName,
|
||||
isDeepseekV4FlashModelId,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isGrokXHighEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isMimoModelIdOrName,
|
||||
isMinimaxM2FamilyModelId,
|
||||
@@ -396,10 +396,10 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
|
||||
return HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
// First-party Grok: `grok-4.20-multi-agent*` advertises `xhigh` (16-agent
|
||||
// mode). Other effort-capable SKUs stay on `minimal/low/medium/high`.
|
||||
// First-party Grok: `grok-4.6*` and `grok-4.20-multi-agent*` advertise
|
||||
// `xhigh`. Other effort-capable SKUs stay on `minimal/low/medium/high`.
|
||||
if (modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "xai")) {
|
||||
return isGrokMultiAgentModelId(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
|
||||
return isGrokXHighEffortCapable(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
return isOpenAICompatReasoningApi(spec.api) &&
|
||||
(isMinimaxM2FamilyModelId(spec.id) ||
|
||||
|
||||
@@ -104468,7 +104468,8 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
@@ -104476,9 +104477,7 @@
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low",
|
||||
"xhigh": "high",
|
||||
"max": "high"
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
@@ -104799,7 +104798,8 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
@@ -104809,9 +104809,7 @@
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low",
|
||||
"xhigh": "high",
|
||||
"max": "high"
|
||||
"minimal": "low"
|
||||
},
|
||||
"includeEncryptedReasoning": true,
|
||||
"filterReasoningHistory": false,
|
||||
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
isGrokModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isGrokXHighEffortCapable,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
isMinimaxM2FamilyModelId,
|
||||
@@ -361,7 +362,25 @@ describe("isGrokMultiAgentModelId", () => {
|
||||
|
||||
test("rejects other Grok ids", () => {
|
||||
expect(isGrokMultiAgentModelId("grok-4.5")).toBe(false);
|
||||
expect(isGrokMultiAgentModelId("grok-4.6")).toBe(false);
|
||||
expect(isGrokMultiAgentModelId("grok-4.20-0309-reasoning")).toBe(false);
|
||||
expect(isGrokMultiAgentModelId("")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isGrokXHighEffortCapable", () => {
|
||||
test("matches grok-4.6 and multi-agent SKUs across namespaces", () => {
|
||||
expect(isGrokXHighEffortCapable("grok-4.6")).toBe(true);
|
||||
expect(isGrokXHighEffortCapable("xai/grok-4.6")).toBe(true);
|
||||
expect(isGrokXHighEffortCapable("xai-oauth/grok-4.6")).toBe(true);
|
||||
expect(isGrokXHighEffortCapable("grok-4.20-multi-agent-0309")).toBe(true);
|
||||
});
|
||||
|
||||
test("rejects Grok SKUs that clamp leftover xhigh to high", () => {
|
||||
expect(isGrokXHighEffortCapable("grok-4.5")).toBe(false);
|
||||
expect(isGrokXHighEffortCapable("grok-4.3")).toBe(false);
|
||||
expect(isGrokXHighEffortCapable("grok-3-mini")).toBe(false);
|
||||
expect(isGrokXHighEffortCapable("grok-build")).toBe(false);
|
||||
expect(isGrokXHighEffortCapable("")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -903,6 +903,26 @@ describe("model thinking runtime helpers", () => {
|
||||
expect(() => requireSupportedEffort(paid, Effort.XHigh)).toThrow(/not supported/);
|
||||
});
|
||||
|
||||
it("exposes xhigh on first-party Grok 4.6 Responses models", () => {
|
||||
const paid = createModel({
|
||||
id: "grok-4.6",
|
||||
api: "openai-responses",
|
||||
provider: "xai",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
});
|
||||
const oauth = createModel({
|
||||
id: "grok-4.6",
|
||||
api: "openai-responses",
|
||||
provider: "xai-oauth",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
});
|
||||
|
||||
expect(paid.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
|
||||
expect(oauth.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
|
||||
expect(requireSupportedEffort(paid, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
expect(paid.compat.reasoningEffortMap?.xhigh).toBeUndefined();
|
||||
});
|
||||
|
||||
it("exposes xhigh on first-party Grok multi-agent Responses models", () => {
|
||||
const paid = createModel({
|
||||
id: "grok-4.20-multi-agent-beta-latest",
|
||||
|
||||
@@ -19,6 +19,14 @@ const XAI_MODELS_DEV_FIXTURE = {
|
||||
limit: { context: 500_000, output: 500_000 },
|
||||
cost: { input: 2, output: 6, cache_read: 0.3 },
|
||||
},
|
||||
"grok-4.6": {
|
||||
name: "Grok 4.6",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text", "image"] },
|
||||
limit: { context: 500_000, output: 500_000 },
|
||||
cost: { input: 2, output: 6, cache_read: 0.5 },
|
||||
},
|
||||
"grok-code-fast-1": {
|
||||
name: "Grok Code Fast 1",
|
||||
tool_call: true,
|
||||
@@ -109,6 +117,18 @@ describe("paid xAI Responses thinking policy", () => {
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
effortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.6"]?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.6"]?.compat).toMatchObject({
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.6"]?.compat).not.toMatchObject({
|
||||
reasoningEffortMap: { xhigh: "high" },
|
||||
});
|
||||
expect(byId["grok-4.20-multi-agent-beta-latest"]?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
@@ -140,6 +160,10 @@ describe("paid xAI Responses thinking policy", () => {
|
||||
expect(bundled["grok-4.5"]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
|
||||
expect(bundled["grok-4.5"]?.thinking?.efforts).not.toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true);
|
||||
expect(bundled["grok-4.6"]?.thinking?.efforts).toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.6"]?.compat).not.toMatchObject({
|
||||
reasoningEffortMap: { xhigh: "high" },
|
||||
});
|
||||
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.thinking?.efforts).toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.compat).not.toMatchObject({
|
||||
reasoningEffortMap: { xhigh: "high" },
|
||||
|
||||
Reference in New Issue
Block a user