fix(catalog): expose grok-4.6 thinking levels on xai-oauth
Add grok-4.6 to the SuperGrok Responses effort allowlist so /model can select low/medium/high/xhigh. Stale omitReasoningEffort cache rows no longer hide the dial. max is omitted because api.x.ai 400s.
This commit is contained in:
@@ -1558,7 +1558,7 @@ xAI Grok OAuth provides subscription-backed access (SuperGrok / X Premium+) to x
|
||||
### Special casings
|
||||
- **Encrypted Reasoning & History Replay**: `includeEncryptedReasoning` is `false` (`packages/catalog/src/compat/openai.ts` `buildOpenAIResponsesCompat`) to suppress encrypted reasoning item replay. `filterReasoningHistory` is `true` (`packages/catalog/src/compat/openai.ts`, `packages/ai/src/providers/openai-responses.ts`) to filter native reasoning items and thinking signatures out of replayed Responses history.
|
||||
- **Image Detail Clamping**: `supportsImageDetailOriginal` is `false` (`packages/catalog/src/compat/openai.ts` `buildOpenAIResponsesCompat`), clamping image detail from `"original"` to `"auto"` because xAI endpoints return HTTP 400/422 on `"original"`.
|
||||
- **Reasoning Effort Gating & Summary**: `supportsReasoningEffort` is `false` unless the model is on the `isGrokReasoningEffortCapable` allowlist (`packages/catalog/src/identity/family.ts`, e.g. `grok-3-mini`, `grok-4.20-multi-agent`, `grok-4.3`, `grok-4.5`). Non-capable models (`grok-build`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, `grok-composer-2.5-fast`) set `omitReasoningEffort: true` to prevent HTTP 400 on `api.x.ai`. `reasoningSummary` is set to `null` (or `undefined` when disabled) in `packages/ai/src/providers/openai-responses.ts` to omit unsupported `reasoning.summary` wire fields.
|
||||
- **Reasoning Effort Gating & Summary**: `supportsReasoningEffort` is `false` unless the model is on the `isGrokReasoningEffortCapable` allowlist (`packages/catalog/src/identity/family.ts`, e.g. `grok-3-mini`, `grok-4.20-multi-agent`, `grok-4.3`, `grok-4.5`, `grok-4.6`). Non-capable models (`grok-build`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, `grok-composer-2.5-fast`) set `omitReasoningEffort: true` to prevent HTTP 400 on `api.x.ai`. `reasoningSummary` is set to `null` (or `undefined` when disabled) in `packages/ai/src/providers/openai-responses.ts` to omit unsupported `reasoning.summary` wire fields.
|
||||
- **Reasoning Effort Map & Caching**: Maps `minimal` to `"low"` (`packages/catalog/src/provider-models/openai-compat.ts` `XAI_REASONING_EFFORT_MAP`). Sends `X-Grok-Conv-Id` for session prompt-cache retention (`promptCacheSessionHeader`).
|
||||
|
||||
### Auth & usage
|
||||
@@ -1566,7 +1566,7 @@ xAI Grok OAuth provides subscription-backed access (SuperGrok / X Premium+) to x
|
||||
- **Usage Tracking**: `xaiOauthUsageProvider` (`packages/ai/src/usage/xai-oauth.ts`) queries `https://cli-chat-proxy.grok.com/v1/billing` (`validateXAIBillingEndpoint` pins to HTTPS `*.grok.com`) with header `X-XAI-Token-Auth: xai-grok-cli` (`getXAICliBillingHeaders`). Only accepts valid OAuth bearer credentials. Probes legacy weekly credits (`?format=credits`, `parseWeeklyBillingConfig` for `creditUsagePercent` and `productUsage`) and unified monthly quota (`parseMonthlyBillingConfig` for `monthlyLimit` and `used`), plus positive `onDemandCap` / `onDemandUsed` limits.
|
||||
|
||||
### Catalog model handling
|
||||
- **Curated Models & Static Seed**: `XAI_OAUTH_CURATED_MODELS` (`packages/catalog/src/provider-models/openai-compat.ts`) defines static models (`grok-build`, `grok-build-0.1`, `grok-4.3`, `grok-4.5`, `grok-4.20-multi-agent-0309`, `grok-4.20-0309-reasoning`, `grok-4.20-0309-non-reasoning`, `grok-composer-2.5-fast`) with zero cost (`cost: 0`). Default model is `grok-4.3` (`descriptors.ts`). `buildXaiOAuthStaticSeed` seeds `ModelRegistry` synchronously at boot so `modelRoles.default = "xai-oauth/<id>"` works before dynamic refresh.
|
||||
- **Curated Models & Static Seed**: `XAI_OAUTH_CURATED_MODELS` (`packages/catalog/src/provider-models/openai-compat.ts`) defines static models (`grok-build`, `grok-build-0.1`, `grok-4.3`, `grok-4.5`, `grok-4.6`, `grok-4.20-multi-agent-0309`, `grok-4.20-0309-reasoning`, `grok-4.20-0309-non-reasoning`, `grok-composer-2.5-fast`) with zero cost (`cost: 0`). Default model is `grok-4.3` (`descriptors.ts`). `buildXaiOAuthStaticSeed` seeds `ModelRegistry` synchronously at boot so `modelRoles.default = "xai-oauth/<id>"` works before dynamic refresh.
|
||||
- **Dynamic Curation Overlay**: `applyXAIOAuthCuration` (`openai-compat.ts`, `xaiOAuthModelManagerOptions`) filters non-chat prefixes (`grok-imagine-`, `grok-stt-`, `grok-voice-`), overlays curated context windows (up to 2M), sets `maxTokens` equal to `contextWindow`, preserves image capabilities and reasoning flags, and injects missing curated models.
|
||||
- **Reference Resolution Exclusion**: `isZeroCostXaiOAuthCandidate` (`packages/catalog/src/identity/reference.ts`) excludes zero-cost subscription entries from reference index matching so subscription pricing and limits do not override public/paid Grok references.
|
||||
|
||||
|
||||
@@ -31,6 +31,19 @@ describe("effort-dial-less reasoner encoding (regression)", () => {
|
||||
expect(getSupportedEfforts(grok43).length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test("xai-oauth/grok-4.6 keeps its effort dial including xhigh", () => {
|
||||
const grok46 = getBundledModel("xai-oauth", "grok-4.6");
|
||||
if (!grok46) throw new Error("xai-oauth/grok-4.6 must be in bundled models.json");
|
||||
expect(grok46.thinking).toBeDefined();
|
||||
expect(getSupportedEfforts(grok46)).toEqual([
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
]);
|
||||
});
|
||||
|
||||
test("xai-oauth/grok-4.20-0309-reasoning reasons but carries no thinking config", () => {
|
||||
const grokR = getBundledModel("xai-oauth", "grok-4.20-0309-reasoning");
|
||||
if (!grokR) throw new Error("xai-oauth/grok-4.20-0309-reasoning must be in bundled models.json");
|
||||
@@ -175,6 +188,16 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
|
||||
encrypted_content: "enc_next_turn",
|
||||
});
|
||||
});
|
||||
|
||||
test("xai-oauth/grok-4.6 sends reasoning.effort xhigh and omits max", () => {
|
||||
const grok46 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.6");
|
||||
if (!grok46) throw new Error("xai-oauth/grok-4.6 must be in bundled models.json");
|
||||
|
||||
const { params } = buildParams(grok46, singleUserContext, { reasoning: Effort.XHigh }, undefined);
|
||||
|
||||
expect(params.reasoning).toEqual({ effort: "xhigh" });
|
||||
expect(getSupportedEfforts(grok46)).not.toContain(Effort.Max);
|
||||
});
|
||||
});
|
||||
|
||||
function followUpContextWithEncryptedReasoning(model: Model<"openai-responses">): Context {
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed SuperGrok (`xai-oauth`) Grok 4.6 hiding the thinking-level picker: the Responses effort-capable allowlist now includes `grok-4.6`, so `/model` can select the documented `low`/`medium`/`high`/`xhigh` ladder (`max` is rejected by api.x.ai).
|
||||
|
||||
## [17.3.5] - 2026-08-16
|
||||
|
||||
### Added
|
||||
@@ -32,6 +36,9 @@
|
||||
- Fixed raw `COPILOT_GITHUB_TOKEN` credentials skipping plan-specific endpoint discovery, which routed GitHub Copilot Business model requests to the personal endpoint and returned HTTP 403. The GitHub Copilot model cache is now scoped per credential, so switching the token no longer serves another account's stale endpoint for the cache TTL ([#8507](https://github.com/can1357/oh-my-pi/issues/8507)).
|
||||
- Fixed the OpenRouter `deepseek/deepseek-v4-pro-0813` route silently clamping the reasoning effort to `high`: the dated SKU advertises (and accepts) the wire-exact `low`/`high`/`max` ladder, so its effort override no longer collapses to `high`-only. The undated `deepseek/deepseek-v4-pro` OpenRouter route stays `high`-only. ([#8517](https://github.com/can1357/oh-my-pi/issues/8517))
|
||||
|
||||
- Fixed SuperGrok (`xai-oauth`) Grok 4.6 hiding the thinking-level picker: the Responses effort-capable allowlist now includes `grok-4.6`, so `/model` can select the documented `low`/`medium`/`high`/`xhigh` ladder (`max` is rejected by api.x.ai).
|
||||
|
||||
- Fixed `streamMarkupHealingPattern` gating DeepSeek DSML healing on a provider-id allowlist, which left DeepSeek models behind user-configured proxies (LiteLLM, private gateways) with no tool-call grammar. Whether the envelope leaks is decided by the serving stack behind the host, not the provider id, so any DeepSeek model on a non-official-OpenAI endpoint now selects `"dsml"`.
|
||||
## [17.3.2] - 2026-08-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -807,6 +807,18 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
|
||||
compat.omitReasoningEffort = true;
|
||||
}
|
||||
// xai-oauth cache/discovery rows written before a SKU joined the
|
||||
// effort-capable allowlist still carry omitReasoningEffort: true. The
|
||||
// allowlist is the live wire contract; do not let that stale flag hide
|
||||
// the picker or strip reasoning.effort.
|
||||
if (
|
||||
spec.provider === "xai-oauth" &&
|
||||
isGrokReasoningEffortCapable(id) &&
|
||||
spec.compat?.supportsReasoningEffort !== false
|
||||
) {
|
||||
compat.supportsReasoningEffort = true;
|
||||
compat.omitReasoningEffort = false;
|
||||
}
|
||||
return compat;
|
||||
}
|
||||
|
||||
|
||||
@@ -121,7 +121,8 @@ const GROK_EFFORT_CAPABLE_PREFIXES = [
|
||||
/**
|
||||
* Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
|
||||
* (e.g. `grok-build`, `grok-4.20-0309-reasoning`) think natively but reject the
|
||||
* param, so callers must omit reasoning effort for them.
|
||||
* param, so callers must omit reasoning effort for them. `grok-4.6` accepts
|
||||
* `low`/`medium`/`high`/`xhigh` and 400s on `max`.
|
||||
*/
|
||||
export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
|
||||
const bare = bareModelId(modelId).trim().toLowerCase();
|
||||
|
||||
@@ -292,6 +292,31 @@ describe("xAI Responses reasoning-effort suppression", () => {
|
||||
expect(buildModel(grokResponsesSpec("grok-code-fast-1", "xai")).thinking).toBeUndefined();
|
||||
});
|
||||
|
||||
it("exposes the grok-4.6 low..xhigh ladder and rejects max", () => {
|
||||
const model = buildModel(grokResponsesSpec("grok-4.6"));
|
||||
expect(model.compat.supportsReasoningEffort).toBe(true);
|
||||
expect(model.compat.omitReasoningEffort).toBe(false);
|
||||
expect(model.thinking?.efforts).toEqual([
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
]);
|
||||
expect(model.thinking?.efforts).not.toContain(Effort.Max);
|
||||
});
|
||||
|
||||
it("lets the grok-4.6 allowlist beat a stale cached omitReasoningEffort flag", () => {
|
||||
const model = buildModel({
|
||||
...grokResponsesSpec("grok-4.6"),
|
||||
compat: { omitReasoningEffort: true },
|
||||
});
|
||||
expect(model.compat.supportsReasoningEffort).toBe(true);
|
||||
expect(model.compat.omitReasoningEffort).toBe(false);
|
||||
expect(model.thinking?.efforts).toContain(Effort.XHigh);
|
||||
expect(model.thinking?.efforts).not.toContain(Effort.Max);
|
||||
});
|
||||
|
||||
it("lets an explicit compat.supportsReasoningEffort override the allowlist default", () => {
|
||||
const compat = buildOpenAIResponsesCompat({
|
||||
...grokResponsesSpec("grok-build"),
|
||||
|
||||
@@ -342,6 +342,7 @@ describe("isGrokReasoningEffortCapable", () => {
|
||||
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.3")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.5")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.6")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("grok-4.6")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("openrouter/xai/grok-3-mini")).toBe(true);
|
||||
});
|
||||
|
||||
|
||||
Reference in New Issue
Block a user