From b49b5b88d233cda1db197567702ddf4fbd704e37 Mon Sep 17 00:00:00 2001 From: Yang Yang Date: Sun, 2 Aug 2026 22:47:15 -0700 Subject: [PATCH] fix(catalog): strip stale xAI Responses effort dials from generated rows Paid xAI models.dev regeneration still emitted Completions-era thinking dials for off-allowlist reasoners. Bake the no-dial policy into the resolver/generator and refresh the exported catalog snapshot. --- packages/ai/CHANGELOG.md | 7 +- packages/catalog/CHANGELOG.md | 23 +- .../catalog/scripts/generated-policies.ts | 5 + packages/catalog/src/models.json | 349 +++++++++++++----- .../src/provider-models/openai-compat.ts | 34 +- .../catalog/test/generated-policies.test.ts | 43 +++ .../xai-responses-thinking-policy.test.ts | 122 ++++++ packages/coding-agent/CHANGELOG.md | 12 +- 8 files changed, 479 insertions(+), 116 deletions(-) create mode 100644 packages/catalog/test/xai-responses-thinking-policy.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index c9810157d..24e74327b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path. + ## [17.3.4] - 2026-08-14 ### Fixed @@ -148,9 +152,6 @@ ### Fixed - Fixed an issue where Ollama requests without a user-role message would fail to generate output or silently fail with a misleading error. -### Fixed - -- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path. ## [17.2.5] - 2026-08-03 diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 5fbe994d8..4abde4e75 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -8,6 +8,14 @@ - Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`. - Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`. - Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter. +- Replayed xAI encrypted reasoning items on later Responses turns instead of stripping `type: "reasoning"` history. + +### Fixed + +- Invalidated stale paid-xAI model-cache rows written under Chat Completions so the Responses migration takes effect immediately instead of waiting for TTL expiry. +- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400. +- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change. +- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners). ## [17.3.4] - 2026-08-14 @@ -134,21 +142,6 @@ - Fixed dynamic discovery for the `deepseek-v4` model family (such as `deepseek-v4-flash-0731`) under `alibaba-token-plan` missing reasoning configuration and maximum thinking effort. - Fixed GitHub Copilot dynamic discovery retaining stale bundled prices for default-context models instead of using the provider's reported default-tier prices. -## [17.2.5] - 2026-08-03 -### Changed - -- Switched the paid xAI provider (`xai` / `XAI_API_KEY`) from Chat Completions to the OpenAI Responses API (`POST https://api.x.ai/v1/responses`), matching SuperGrok `xai-oauth`. Prompt-cache affinity (`x-grok-conv-id`), reasoning-effort allowlisting, and encrypted-reasoning replay rules are now shared across both first-party xAI hosts. -- Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`. -- Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`. -- Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter. -- Replayed xAI encrypted reasoning items on later Responses turns instead of stripping `type: "reasoning"` history. - -### Fixed - -- Invalidated stale paid-xAI model-cache rows written under Chat Completions so the Responses migration takes effect immediately instead of waiting for TTL expiry. -- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400. -- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change. - ## [17.2.5] - 2026-08-03 ### Fixed diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 2b97e0dfb..0002dab3e 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -22,6 +22,7 @@ import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../sr import { ALIBABA_TOKEN_PLAN_STATIC_MODELS, OPENAI_GPT_56_LONG_CONTEXT_COSTS, + applyXaiResponsesThinkingPolicy, resolveWaferServerlessThinkingFormat, } from "../src/provider-models/openai-compat"; import type { Api, LongContextTokenCost, Model, ModelSpec } from "../src/types"; @@ -353,6 +354,10 @@ export function applyOllamaCloudOutputCap(models: ModelSpec[]): void { } function applyGeneratedModelPolicy(model: ModelSpec): void { + if (model.provider === "xai" && model.api === "openai-responses") { + const updated = applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">); + model.compat = updated.compat; + } const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined; if (copilotLimits) { model.contextWindow = copilotLimits.contextWindow; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 7bd63429a..c28196f6c 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -103730,7 +103730,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-2-1212": { "id": "grok-2-1212", @@ -103749,7 +103756,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-2-latest": { "id": "grok-2-latest", @@ -103768,7 +103782,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-2-vision": { "id": "grok-2-vision", @@ -103788,7 +103809,14 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": 4096 + "maxTokens": 4096, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-2-vision-1212": { "id": "grok-2-vision-1212", @@ -103808,7 +103836,14 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": 4096 + "maxTokens": 4096, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-2-vision-latest": { "id": "grok-2-vision-latest", @@ -103828,7 +103863,14 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": 4096 + "maxTokens": 4096, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-3": { "id": "grok-3", @@ -103847,7 +103889,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-3-fast": { "id": "grok-3-fast", @@ -103866,7 +103915,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-3-fast-latest": { "id": "grok-3-fast-latest", @@ -103885,7 +103941,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-3-latest": { "id": "grok-3-latest", @@ -103904,7 +103967,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-3-mini": { "id": "grok-3-mini", @@ -103930,8 +104000,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-3-mini-fast": { @@ -103958,8 +104039,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-3-mini-fast-latest": { @@ -103986,8 +104078,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-3-mini-latest": { @@ -104014,8 +104117,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-4": { @@ -104036,14 +104150,12 @@ }, "contextWindow": 256000, "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-4-1-fast": { @@ -104065,14 +104177,12 @@ }, "contextWindow": 2000000, "maxTokens": 30000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-4-1-fast-non-reasoning": { @@ -104093,7 +104203,14 @@ "cacheWrite": 0 }, "contextWindow": 2000000, - "maxTokens": 30000 + "maxTokens": 30000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-4-fast": { "id": "grok-4-fast", @@ -104114,14 +104231,12 @@ }, "contextWindow": 2000000, "maxTokens": 30000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-4-fast-non-reasoning": { @@ -104142,7 +104257,14 @@ "cacheWrite": 0 }, "contextWindow": 2000000, - "maxTokens": 30000 + "maxTokens": 30000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-4.20-0309-non-reasoning": { "id": "grok-4.20-0309-non-reasoning", @@ -104162,7 +104284,14 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 30000 + "maxTokens": 30000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-4.20-0309-reasoning": { "id": "grok-4.20-0309-reasoning", @@ -104183,15 +104312,12 @@ }, "contextWindow": 1000000, "maxTokens": 30000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-4.20-beta-latest-non-reasoning": { @@ -104212,7 +104338,14 @@ "cacheWrite": 0 }, "contextWindow": 2000000, - "maxTokens": 30000 + "maxTokens": 30000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-4.20-beta-latest-reasoning": { "id": "grok-4.20-beta-latest-reasoning", @@ -104233,15 +104366,12 @@ }, "contextWindow": 2000000, "maxTokens": 30000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "requiresEffort": true + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-4.20-multi-agent-beta-latest": { @@ -104269,8 +104399,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-4.3": { @@ -104298,8 +104439,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-4.5": { @@ -104327,8 +104479,19 @@ "minimal", "low", "medium", - "high" - ] + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": true, + "omitReasoningEffort": false } }, "grok-4.6": { @@ -104377,7 +104540,14 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 4096 + "maxTokens": 4096, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } }, "grok-build-0.1": { "id": "grok-build-0.1", @@ -104398,14 +104568,12 @@ }, "contextWindow": 256000, "maxTokens": 256000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-code-fast-1": { @@ -104426,14 +104594,12 @@ }, "contextWindow": 256000, "maxTokens": 10000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true } }, "grok-vision-beta": { @@ -104454,7 +104620,14 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": 4096 + "maxTokens": 4096, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false, + "omitReasoningEffort": true + } } }, "xai-oauth": { diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 8c9fe3384..1edf057e3 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1376,6 +1376,36 @@ function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): Model // of the omitReasoningEffort gate in pi-ai's stream.ts. const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const; +/** + * Bake first-party xAI Responses effort-dial metadata onto a catalog spec. + * + * models.dev marks many Grok SKUs as reasoners and the thinking rebake would + * otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only + * accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids — + * off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`, + * `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok + * (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid + * `xai` rows come from stencil.so and need the same wire facts in the exported + * `models.json` so direct catalog readers do not present an unsupported dial. + * + * Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win. + */ +export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> { + const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id); + return { + ...model, + compat: { + ...(model.compat ?? {}), + reasoningEffortMap: { + ...XAI_REASONING_EFFORT_MAP, + ...(model.compat?.reasoningEffortMap ?? {}), + }, + supportsReasoningEffort: effortCapable, + omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable, + }, + }; +} + // xai-oauth's /v1/models exposes no per-request output limit on the OAuth // (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens` // like it owns `contextWindow`: each entry mirrors its context window. The @@ -5854,7 +5884,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor defaultContextWindow: 131072, }), // --- xAI --- - openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1"), + openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", { + transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">), + }), // --- DeepSeek --- openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", { // Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 3c88063ba..f7ae34900 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -467,6 +467,49 @@ describe("generated model policies", () => { expect(models[2]?.applyPatchToolType).toBeUndefined(); expect(models[3]?.applyPatchToolType).toBeUndefined(); }); + + it("strips paid xAI Responses effort dials for off-allowlist reasoners", () => { + const models: ModelSpec<"openai-responses">[] = [ + createSpec({ + id: "grok-code-fast-1", + api: "openai-responses", + provider: "xai", + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + }), + createSpec({ + id: "grok-4.5", + api: "openai-responses", + provider: "xai", + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + }), + createSpec({ + id: "grok-code-fast-1", + api: "openai-responses", + provider: "openrouter", + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + }), + ]; + + applyGeneratedModelPolicies(models); + + expect(models[0]?.thinking).toBeUndefined(); + expect(models[0]?.compat).toMatchObject({ + supportsReasoningEffort: false, + omitReasoningEffort: true, + reasoningEffortMap: { minimal: "low" }, + }); + expect(models[1]?.thinking?.efforts).toEqual([ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + ]); + expect(models[1]?.compat?.supportsReasoningEffort).toBe(true); + // Non-xAI hosts are outside this policy — no baked no-dial compat. + expect(models[2]?.thinking).toBeDefined(); + expect(models[2]?.compat?.supportsReasoningEffort).toBeUndefined(); + }); }); describe("applyOllamaCloudOutputCap", () => { diff --git a/packages/catalog/test/xai-responses-thinking-policy.test.ts b/packages/catalog/test/xai-responses-thinking-policy.test.ts new file mode 100644 index 000000000..ed51571ed --- /dev/null +++ b/packages/catalog/test/xai-responses-thinking-policy.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import MODELS_JSON from "@oh-my-pi/pi-catalog/models.json" with { type: "json" }; +import { + MODELS_DEV_PROVIDER_DESCRIPTORS, + mapModelsDevToModels, +} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; +import { applyGeneratedModelPolicies } from "../scripts/generated-policies"; + +const XAI_MODELS_DEV_FIXTURE = { + xai: { + models: { + "grok-4.5": { + name: "Grok 4.5", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image"] }, + limit: { context: 500_000, output: 500_000 }, + cost: { input: 2, output: 6, cache_read: 0.3 }, + }, + "grok-code-fast-1": { + name: "Grok Code Fast 1", + tool_call: true, + reasoning: true, + modalities: { input: ["text"] }, + limit: { context: 256_000, output: 10_000 }, + cost: { input: 0.2, output: 1.5 }, + }, + "grok-build-0.1": { + name: "Grok Build 0.1", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image"] }, + limit: { context: 256_000, output: 256_000 }, + cost: { input: 0, output: 0 }, + }, + "grok-4.20-0309-reasoning": { + name: "Grok 4.20 (Reasoning)", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image"] }, + limit: { context: 2_000_000, output: 64_000 }, + cost: { input: 2, output: 6 }, + }, + "grok-2": { + name: "Grok 2", + tool_call: true, + reasoning: false, + modalities: { input: ["text"] }, + limit: { context: 131_072, output: 8192 }, + cost: { input: 2, output: 10 }, + }, + }, + }, +}; + +describe("paid xAI Responses thinking policy", () => { + it("bakes the effort-dial allowlist on stencil.so → openai-responses mapping", () => { + const mapped = mapModelsDevToModels(XAI_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( + model => model.provider === "xai", + ); + const byId = Object.fromEntries(mapped.map(model => [model.id, model])); + + expect(byId["grok-4.5"]?.api).toBe("openai-responses"); + expect(byId["grok-4.5"]?.compat).toMatchObject({ + supportsReasoningEffort: true, + omitReasoningEffort: false, + reasoningEffortMap: { minimal: "low" }, + }); + for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) { + expect(byId[id]?.reasoning, id).toBe(true); + expect(byId[id]?.compat, id).toMatchObject({ + supportsReasoningEffort: false, + omitReasoningEffort: true, + reasoningEffortMap: { minimal: "low" }, + }); + } + expect(byId["grok-2"]?.compat).toMatchObject({ + supportsReasoningEffort: false, + omitReasoningEffort: true, + }); + }); + + it("strips stale thinking dials from off-allowlist paid xAI reasoners during generation", () => { + const mapped = mapModelsDevToModels(XAI_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( + model => model.provider === "xai", + ); + // Snapshot-era Completions rows still carry a default effort ladder after the + // api flip; the generator must not re-emit that dial for Responses. + const snapshotStale = mapped.find(model => model.id === "grok-code-fast-1"); + expect(snapshotStale).toBeDefined(); + snapshotStale!.thinking = { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }; + + applyGeneratedModelPolicies(mapped); + const byId = Object.fromEntries(mapped.map(model => [model.id, model])); + + expect(byId["grok-4.5"]?.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { minimal: "low" }, + }); + for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) { + expect(byId[id]?.reasoning, id).toBe(true); + expect(byId[id]?.thinking, id).toBeUndefined(); + expect(byId[id]?.compat, id).toMatchObject({ supportsReasoningEffort: false }); + } + }); + + it("exports no-dial rows in the bundled models.json snapshot", () => { + const bundled = + (MODELS_JSON as unknown as Record>>).xai ?? {}; + for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) { + expect(bundled[id], `xai/${id} missing from models.json`).toBeDefined(); + expect(bundled[id]?.reasoning, id).toBe(true); + expect(bundled[id]?.thinking, id).toBeUndefined(); + expect(bundled[id]?.compat?.supportsReasoningEffort, id).toBe(false); + } + expect(bundled["grok-4.5"]?.thinking?.efforts).toContain(Effort.XHigh); + expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 61ccd6e52..829b041b2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -8,6 +8,9 @@ - Changed the default model for `XAI_API_KEY` (`xai`) from `grok-4-fast-non-reasoning` to `grok-4.5`. - Changed the default model for SuperGrok OAuth (`xai-oauth`) from `grok-4.3` to `grok-4.5`. - Included `reasoning.encrypted_content` in Responses `include` for paid xAI and SuperGrok OAuth models. +- Replayed encrypted xAI reasoning on follow-up Responses turns for `xai` and `xai-oauth`. +- Kept automatic model selection on paid `xai/grok-4.5` when only `XAI_API_KEY` is set, instead of preferring SuperGrok `xai-oauth/grok-4.5`. +- Stopped sending presence/frequency penalties and stop sequences to xAI reasoning models such as `grok-4.5`, which reject them. ## [17.3.4] - 2026-08-14 @@ -364,15 +367,6 @@ - Fixed issues with `/btw` branch promotion where branches could park behind active turns, cut from outdated session leaves, or leave rejected branch keys indistinguishable from composer input. - Fixed database bloat by ensuring archived main and nested session rows are properly cleaned up from `stats.db` during garbage collection. - Fixed startup hanging during local model discovery when a timed-out transport left its request pending, which blocked the CLI before OAuth login could finish ([#7482](https://github.com/can1357/oh-my-pi/issues/7482)). -### Changed - -- Routed paid xAI models (`XAI_API_KEY` / `xai/…`) through the Responses API used by SuperGrok OAuth instead of Chat Completions. -- Changed the default model for `XAI_API_KEY` (`xai`) from `grok-4-fast-non-reasoning` to `grok-4.5`. -- Changed the default model for SuperGrok OAuth (`xai-oauth`) from `grok-4.3` to `grok-4.5`. -- Included `reasoning.encrypted_content` in Responses `include` for paid xAI and SuperGrok OAuth models. -- Replayed encrypted xAI reasoning on follow-up Responses turns for `xai` and `xai-oauth`. -- Kept automatic model selection on paid `xai/grok-4.5` when only `XAI_API_KEY` is set, instead of preferring SuperGrok `xai-oauth/grok-4.5`. -- Stopped sending presence/frequency penalties and stop sequences to xAI reasoning models such as `grok-4.5`, which reject them. ## [17.2.5] - 2026-08-03