fix(catalog): drop unsupported xhigh effort from first-party Grok

api.x.ai accepts low/medium/high (and clamps minimal to low). Stop
advertising xhigh on paid xai and SuperGrok Responses rows, and map
leftover xhigh/max requests to high.
This commit is contained in:
Yang Yang
2026-08-09 22:34:42 -07:00
parent 6c0f458279
commit 09830d2bd6
10 changed files with 141 additions and 71 deletions
+1 -1
View File
@@ -5,6 +5,7 @@
### Fixed
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
## [17.3.4] - 2026-08-14
@@ -90,7 +91,6 @@
- Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
- Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
- Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)).
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
## [17.2.12] - 2026-08-08
+2 -1
View File
@@ -16,6 +16,8 @@
- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400.
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
- Removed unsupported `xhigh` (and `max`) thinking tiers from first-party Grok Responses catalog rows; leftover `xhigh`/`max` requests clamp to `high`.
## [17.3.4] - 2026-08-14
@@ -93,7 +95,6 @@
- Marked `meta/muse-spark-1.2` and `muse-spark-1.2-contributor` as image-capable (`input: ["text", "image"]`) with the same Responses reasoning, thinking, and cost metadata as `muse-spark-1.1` (contributor uses its discounted 0.1/0.2 pricing), so `omp models` no longer lists them as text-only.
- Fixed GLM-5.2 thinking levels across Baseten, CoreWeave, HuggingFace, and other uppercase-ID resellers, which were getting the generic `xhigh` effort ladder instead of the GLM-5.2-specific tiers. Also added Baseten `zai-org/GLM-5.2-Fast` and Fireworks `glm-5.2-fast` as reasoning models ([#8200](https://github.com/can1357/oh-my-pi/pull/8200) by [@jcfrancisco](https://github.com/jcfrancisco)).
- Marked first-party xAI Responses hosts (`xai` and `xai-oauth`) as not supporting `reasoning.summary`, so paid `xai/grok-4.5` effort requests omit the unsupported field instead of sending `summary: "auto"`.
## [17.2.12] - 2026-08-08
+3 -1
View File
@@ -177,9 +177,11 @@ const MIMO_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]>
xhigh: "high",
};
/** xAI `/v1/responses` accepts `low|medium|high` (and `xhigh` on some SKUs), not `minimal`. */
/** xAI `/v1/responses` accepts `low|medium|high`, not `minimal`/`xhigh`/`max`. */
const XAI_RESPONSES_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
minimal: "low",
xhigh: "high",
max: "high",
};
function mergeModelReasoningEffortMap(
+6
View File
@@ -395,6 +395,12 @@ function getModelDefinedEfforts<TApi extends Api>(
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
return HIGH_MAX_REASONING_EFFORTS;
}
// api.x.ai accepts `low|medium|high` (and clamps `minimal` → `low`). It
// rejects `xhigh`/`max`, so first-party Grok Responses rows must not
// advertise those tiers.
if (modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "xai")) {
return DEFAULT_REASONING_EFFORTS;
}
return isOpenAICompatReasoningApi(spec.api) &&
(isMinimaxM2FamilyModelId(spec.id) ||
isOpenAIGptOssModelId(spec.id) ||
+100 -50
View File
@@ -103733,7 +103733,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103759,7 +103761,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103785,7 +103789,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103812,7 +103818,9 @@
"maxTokens": 4096,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103839,7 +103847,9 @@
"maxTokens": 4096,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103866,7 +103876,9 @@
"maxTokens": 4096,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103892,7 +103904,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103918,7 +103932,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103944,7 +103960,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -103970,7 +103988,9 @@
"maxTokens": 8192,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104000,8 +104020,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104009,7 +104028,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104039,8 +104060,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104048,7 +104068,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104078,8 +104100,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104087,7 +104108,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104117,8 +104140,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104126,7 +104148,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104152,7 +104176,9 @@
"maxTokens": 64000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104179,7 +104205,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104206,7 +104234,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104233,7 +104263,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104260,7 +104292,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104287,7 +104321,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104314,7 +104350,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104341,7 +104379,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104368,7 +104408,9 @@
"maxTokens": 30000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104399,8 +104441,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104408,7 +104449,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104439,8 +104482,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104448,7 +104490,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104479,8 +104523,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104488,7 +104531,9 @@
},
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": true,
"omitReasoningEffort": false
@@ -104543,7 +104588,9 @@
"maxTokens": 4096,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104570,7 +104617,9 @@
"maxTokens": 256000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104596,7 +104645,9 @@
"maxTokens": 10000,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104623,7 +104674,9 @@
"maxTokens": 4096,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
"minimal": "low",
"xhigh": "high",
"max": "high"
},
"supportsReasoningEffort": false,
"omitReasoningEffort": true
@@ -104719,8 +104772,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104764,8 +104816,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -104809,8 +104860,7 @@
"minimal",
"low",
"medium",
"high",
"xhigh"
"high"
],
"effortMap": {
"minimal": "low"
@@ -1375,13 +1375,10 @@ function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): Model
return { ...model, compat };
}
// Hermes-agent parity: only the `minimal -> low` clamp is applied (see
// hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
// "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
// — let xAI decide if the level is valid for the specific Grok model.
// `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
// of the omitReasoningEffort gate in pi-ai's stream.ts.
const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
// Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
// codex.py:92). api.x.ai also rejects `xhigh`/`max`, so those clamp to `high`.
// `resolveModelThinking` folds this into `model.thinking.effortMap`.
const XAI_REASONING_EFFORT_MAP = { minimal: "low", xhigh: "high", max: "high" } as const;
/**
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
+2 -2
View File
@@ -264,8 +264,8 @@ describe("xAI Responses reasoning-effort suppression", () => {
expect(oauth.supportsImageDetailOriginal).toBe(false);
expect(paid.supportsReasoningEffort).toBe(true);
expect(oauth.supportsReasoningEffort).toBe(true);
expect(paid.reasoningEffortMap).toEqual({ minimal: "low" });
expect(oauth.reasoningEffortMap).toEqual({ minimal: "low" });
expect(paid.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
expect(oauth.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
expect(paid.supportsPenaltyAndStopParams).toBe(false);
expect(oauth.supportsPenaltyAndStopParams).toBe(false);
expect(paid.supportsReasoningSummary).toBe(false);
@@ -498,13 +498,7 @@ describe("generated model policies", () => {
omitReasoningEffort: true,
reasoningEffortMap: { minimal: "low" },
});
expect(models[1]?.thinking?.efforts).toEqual([
Effort.Minimal,
Effort.Low,
Effort.Medium,
Effort.High,
Effort.XHigh,
]);
expect(models[1]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(models[1]?.compat?.supportsReasoningEffort).toBe(true);
// Non-xAI hosts are outside this policy — no baked no-dial compat.
expect(models[2]?.thinking).toBeDefined();
@@ -884,6 +884,25 @@ describe("model thinking runtime helpers", () => {
expect(() => requireSupportedEffort(opus46, Effort.XHigh)).toThrow(/not supported/);
});
it("does not expose xhigh on first-party xAI Grok Responses models", () => {
const paid = createModel({
id: "grok-4.5",
api: "openai-responses",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
});
const oauth = createModel({
id: "grok-4.5",
api: "openai-responses",
provider: "xai-oauth",
baseUrl: "https://api.x.ai/v1",
});
expect(paid.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(oauth.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(() => requireSupportedEffort(paid, Effort.XHigh)).toThrow(/not supported/);
});
it("rejects effort requests against un-built reasoning specs", () => {
const spec = {
id: "broken-reasoner",
@@ -97,7 +97,7 @@ describe("paid xAI Responses thinking policy", () => {
expect(byId["grok-4.5"]?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
effortMap: { minimal: "low" },
});
for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) {
@@ -116,7 +116,8 @@ describe("paid xAI Responses thinking policy", () => {
expect(bundled[id]?.thinking, id).toBeUndefined();
expect(bundled[id]?.compat?.supportsReasoningEffort, id).toBe(false);
}
expect(bundled["grok-4.5"]?.thinking?.efforts).toContain(Effort.XHigh);
expect(bundled["grok-4.5"]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(bundled["grok-4.5"]?.thinking?.efforts).not.toContain(Effort.XHigh);
expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true);
});
});