fix(catalog): keep xhigh on Grok multi-agent Responses models
grok-4.20-multi-agent uses reasoning.effort for agent count, and xhigh is the 16-agent mode. Leave that tier advertised and unmapped while Grok 4.5 still clamps leftover xhigh/max to high.
This commit is contained in:
@@ -17,7 +17,7 @@
|
||||
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
|
||||
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
|
||||
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
|
||||
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok Responses catalog rows; leftover `xhigh`/`max` requests clamp to `high`.
|
||||
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok 4.5 / 4.3 / 3-mini Responses rows; leftover `xhigh`/`max` requests clamp to `high`. `grok-4.20-multi-agent*` still advertises unmapped `xhigh` (16-agent mode).
|
||||
- Stopped baking `reasoningEffortMap` on first-party xAI catalog rows that omit `reasoning.effort` (`omitReasoningEffort: true`).
|
||||
- Suppressed presence/frequency penalties on every first-party xAI Responses model, including non-reasoning ids such as `grok-2`; xAI's `/v1/responses` marks those fields unsupported.
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
isClaudeModelId,
|
||||
isDeepseekModelIdOrName,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isKimiK26ModelId,
|
||||
@@ -177,13 +178,22 @@ const MIMO_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]>
|
||||
xhigh: "high",
|
||||
};
|
||||
|
||||
/** xAI `/v1/responses` accepts `low|medium|high`, not `minimal`/`xhigh`/`max`. */
|
||||
const XAI_RESPONSES_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
|
||||
/** Shared `minimal → low` clamp. Multi-agent Grok keeps `xhigh` unmapped. */
|
||||
const XAI_RESPONSES_MINIMAL_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
|
||||
minimal: "low",
|
||||
};
|
||||
/** Non-multi-agent Grok: leftover `xhigh`/`max` clamp to `high` (4.5 has no 16-agent mode). */
|
||||
const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
|
||||
minimal: "low",
|
||||
xhigh: "high",
|
||||
max: "high",
|
||||
};
|
||||
|
||||
/** Wire effort remap for first-party xAI Responses. */
|
||||
export function xaiResponsesReasoningEffortMap(modelId: string): NonNullable<OpenAICompat["reasoningEffortMap"]> {
|
||||
return isGrokMultiAgentModelId(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
|
||||
}
|
||||
|
||||
function mergeModelReasoningEffortMap(
|
||||
compat: ResolvedOpenAISharedCompat,
|
||||
modelId: string,
|
||||
@@ -717,7 +727,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
!isXaiHost && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
|
||||
// api.x.ai rejects `reasoning.summary` (SuperGrok and paid key alike).
|
||||
supportsReasoningSummary: !isXaiHost,
|
||||
reasoningEffortMap: isXaiHost ? { ...XAI_RESPONSES_REASONING_EFFORT_MAP } : {},
|
||||
reasoningEffortMap: isXaiHost ? { ...xaiResponsesReasoningEffortMap(id) } : {},
|
||||
supportsReasoningParams: true,
|
||||
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
|
||||
// temperature/top_p/… with a 400 on every serving host (#5606).
|
||||
@@ -781,7 +791,15 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
};
|
||||
applyCompatOverrides(compat, spec.compat);
|
||||
if (isXaiHost) {
|
||||
compat.reasoningEffortMap = { ...XAI_RESPONSES_REASONING_EFFORT_MAP, ...compat.reasoningEffortMap };
|
||||
const canonical = xaiResponsesReasoningEffortMap(id);
|
||||
compat.reasoningEffortMap = { ...compat.reasoningEffortMap, ...canonical };
|
||||
// Multi-agent Grok advertises unmapped `xhigh`; drop a stale clamp from
|
||||
// previous snapshots so 16-agent mode is not rewritten to `high`.
|
||||
for (const key of ["xhigh", "max"] as const) {
|
||||
if (!(key in canonical)) {
|
||||
delete compat.reasoningEffortMap[key];
|
||||
}
|
||||
}
|
||||
}
|
||||
if (spec.compat?.reasoningDisableMode === undefined) {
|
||||
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
|
||||
|
||||
@@ -123,6 +123,15 @@ export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
|
||||
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
|
||||
});
|
||||
|
||||
/**
|
||||
* `grok-4.20-multi-agent*` uses `reasoning.effort` to pick agent count
|
||||
* (`xhigh` is the 16-agent mode). Other first-party Grok effort SKUs stay on
|
||||
* `low|medium|high` (https://docs.x.ai/developers/model-capabilities/text/reasoning).
|
||||
*/
|
||||
export const isGrokMultiAgentModelId = memo((modelId: string): boolean => {
|
||||
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.20-multi-agent");
|
||||
});
|
||||
|
||||
/**
|
||||
* MiniMax M2-generation family (M2, M2.1, M2.5, M2.7, including `-highspeed`/
|
||||
* `-lightning`/`-her`/`-turbo` variants, dotless aliases like `minimax-m21`,
|
||||
|
||||
@@ -26,6 +26,7 @@ import {
|
||||
isDeepseekModelIdOrName,
|
||||
isDeepseekV4FlashModelId,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isKimiK3ModelId,
|
||||
isMimoModelIdOrName,
|
||||
isMinimaxM2FamilyModelId,
|
||||
@@ -395,11 +396,10 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
|
||||
return HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
// api.x.ai accepts `low|medium|high` (and clamps `minimal` → `low`). It
|
||||
// rejects `xhigh`/`max`, so first-party Grok Responses rows must not
|
||||
// advertise those tiers.
|
||||
// First-party Grok: `grok-4.20-multi-agent*` advertises `xhigh` (16-agent
|
||||
// mode). Other effort-capable SKUs stay on `minimal/low/medium/high`.
|
||||
if (modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "xai")) {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
return isGrokMultiAgentModelId(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
return isOpenAICompatReasoningApi(spec.api) &&
|
||||
(isMinimaxM2FamilyModelId(spec.id) ||
|
||||
|
||||
@@ -104346,7 +104346,8 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
@@ -104354,9 +104355,7 @@
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low",
|
||||
"xhigh": "high",
|
||||
"max": "high"
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
@@ -104651,7 +104650,8 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
@@ -104661,9 +104661,7 @@
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low",
|
||||
"xhigh": "high",
|
||||
"max": "high"
|
||||
"minimal": "low"
|
||||
},
|
||||
"includeEncryptedReasoning": true,
|
||||
"filterReasoningHistory": false,
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { USER_AGENT } from "@oh-my-pi/pi-utils";
|
||||
import * as logger from "@oh-my-pi/pi-utils/logger";
|
||||
import { xaiResponsesReasoningEffortMap } from "../compat/openai";
|
||||
import {
|
||||
DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
|
||||
fetchOpenAICompatibleModels,
|
||||
@@ -1376,9 +1377,9 @@ function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): Model
|
||||
}
|
||||
|
||||
// Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
|
||||
// codex.py:92). api.x.ai also rejects `xhigh`/`max`, so those clamp to `high`.
|
||||
// codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
|
||||
// other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
|
||||
// `resolveModelThinking` folds this into `model.thinking.effortMap`.
|
||||
const XAI_REASONING_EFFORT_MAP = { minimal: "low", xhigh: "high", max: "high" } as const;
|
||||
|
||||
/**
|
||||
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
|
||||
@@ -1402,10 +1403,7 @@ export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-respons
|
||||
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
|
||||
};
|
||||
if (effortCapable) {
|
||||
compat.reasoningEffortMap = {
|
||||
...XAI_REASONING_EFFORT_MAP,
|
||||
...(model.compat?.reasoningEffortMap ?? {}),
|
||||
};
|
||||
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
|
||||
} else {
|
||||
delete compat.reasoningEffortMap;
|
||||
}
|
||||
@@ -1425,7 +1423,7 @@ export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-respons
|
||||
// reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
|
||||
// false). Caller supplies a `base` Model (either a freshly synthesised seed
|
||||
// or a dynamic-fetched entry); the helper layers curated fields on top.
|
||||
// The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is merged
|
||||
// The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
|
||||
// only onto effort-capable rows. Off-allowlist reasoners omit the wire
|
||||
// param, so a map on those specs is dead weight.
|
||||
// The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
|
||||
@@ -1445,7 +1443,7 @@ function mergeCuratedIntoModel(
|
||||
supportsReasoningEffort: effortCapable,
|
||||
};
|
||||
if (effortCapable) {
|
||||
compat.reasoningEffortMap = { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) };
|
||||
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
|
||||
} else {
|
||||
delete compat.reasoningEffortMap;
|
||||
}
|
||||
@@ -1550,7 +1548,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: curated.contextWindow,
|
||||
maxTokens: curated.contextWindow,
|
||||
compat: { reasoningEffortMap: XAI_REASONING_EFFORT_MAP },
|
||||
compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
|
||||
};
|
||||
return mergeCuratedIntoModel(base, curated);
|
||||
});
|
||||
|
||||
@@ -266,6 +266,9 @@ describe("xAI Responses reasoning-effort suppression", () => {
|
||||
expect(oauth.supportsReasoningEffort).toBe(true);
|
||||
expect(paid.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
|
||||
expect(oauth.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
|
||||
expect(
|
||||
buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.20-multi-agent-0309", "xai")).reasoningEffortMap,
|
||||
).toEqual({ minimal: "low" });
|
||||
expect(paid.supportsPenaltyAndStopParams).toBe(false);
|
||||
expect(oauth.supportsPenaltyAndStopParams).toBe(false);
|
||||
expect(paid.supportsReasoningSummary).toBe(false);
|
||||
|
||||
@@ -5,6 +5,7 @@ import {
|
||||
isGeminiModelId,
|
||||
isGlmVisionModelId,
|
||||
isGrokModelId,
|
||||
isGrokMultiAgentModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
@@ -349,3 +350,17 @@ describe("isGrokReasoningEffortCapable", () => {
|
||||
expect(isGrokReasoningEffortCapable("")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isGrokMultiAgentModelId", () => {
|
||||
test("matches grok-4.20-multi-agent SKUs across namespaces", () => {
|
||||
expect(isGrokMultiAgentModelId("grok-4.20-multi-agent")).toBe(true);
|
||||
expect(isGrokMultiAgentModelId("grok-4.20-multi-agent-0309")).toBe(true);
|
||||
expect(isGrokMultiAgentModelId("xai/grok-4.20-multi-agent-beta-latest")).toBe(true);
|
||||
});
|
||||
|
||||
test("rejects other Grok ids", () => {
|
||||
expect(isGrokMultiAgentModelId("grok-4.5")).toBe(false);
|
||||
expect(isGrokMultiAgentModelId("grok-4.20-0309-reasoning")).toBe(false);
|
||||
expect(isGrokMultiAgentModelId("")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -884,7 +884,7 @@ describe("model thinking runtime helpers", () => {
|
||||
expect(() => requireSupportedEffort(opus46, Effort.XHigh)).toThrow(/not supported/);
|
||||
});
|
||||
|
||||
it("does not expose xhigh on first-party xAI Grok Responses models", () => {
|
||||
it("does not expose xhigh on first-party Grok 4.5 Responses models", () => {
|
||||
const paid = createModel({
|
||||
id: "grok-4.5",
|
||||
api: "openai-responses",
|
||||
@@ -903,6 +903,26 @@ describe("model thinking runtime helpers", () => {
|
||||
expect(() => requireSupportedEffort(paid, Effort.XHigh)).toThrow(/not supported/);
|
||||
});
|
||||
|
||||
it("exposes xhigh on first-party Grok multi-agent Responses models", () => {
|
||||
const paid = createModel({
|
||||
id: "grok-4.20-multi-agent-beta-latest",
|
||||
api: "openai-responses",
|
||||
provider: "xai",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
});
|
||||
const oauth = createModel({
|
||||
id: "grok-4.20-multi-agent-0309",
|
||||
api: "openai-responses",
|
||||
provider: "xai-oauth",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
});
|
||||
|
||||
expect(paid.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
|
||||
expect(oauth.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
|
||||
expect(requireSupportedEffort(paid, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
expect(paid.compat.reasoningEffortMap?.xhigh).toBeUndefined();
|
||||
});
|
||||
|
||||
it("rejects effort requests against un-built reasoning specs", () => {
|
||||
const spec = {
|
||||
id: "broken-reasoner",
|
||||
|
||||
@@ -51,6 +51,14 @@ const XAI_MODELS_DEV_FIXTURE = {
|
||||
limit: { context: 131_072, output: 8192 },
|
||||
cost: { input: 2, output: 10 },
|
||||
},
|
||||
"grok-4.20-multi-agent-beta-latest": {
|
||||
name: "Grok 4.20 (Multi-Agent)",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text"] },
|
||||
limit: { context: 2_000_000, output: 64_000 },
|
||||
cost: { input: 2, output: 6 },
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -101,6 +109,18 @@ describe("paid xAI Responses thinking policy", () => {
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
effortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.20-multi-agent-beta-latest"]?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.20-multi-agent-beta-latest"]?.compat).toMatchObject({
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortMap: { minimal: "low" },
|
||||
});
|
||||
expect(byId["grok-4.20-multi-agent-beta-latest"]?.compat).not.toMatchObject({
|
||||
reasoningEffortMap: { xhigh: "high" },
|
||||
});
|
||||
for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) {
|
||||
expect(byId[id]?.reasoning, id).toBe(true);
|
||||
expect(byId[id]?.thinking, id).toBeUndefined();
|
||||
@@ -120,5 +140,9 @@ describe("paid xAI Responses thinking policy", () => {
|
||||
expect(bundled["grok-4.5"]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
|
||||
expect(bundled["grok-4.5"]?.thinking?.efforts).not.toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true);
|
||||
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.thinking?.efforts).toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.20-multi-agent-beta-latest"]?.compat).not.toMatchObject({
|
||||
reasoningEffortMap: { xhigh: "high" },
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user