fix(catalog): strip stale xAI Responses effort dials from generated rows
Paid xAI models.dev regeneration still emitted Completions-era thinking dials for off-allowlist reasoners. Bake the no-dial policy into the resolver/generator and refresh the exported catalog snapshot.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
### Fixed
|
||||
@@ -148,9 +152,6 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed an issue where Ollama requests without a user-role message would fail to generate output or silently fail with a misleading error.
|
||||
### Fixed
|
||||
|
||||
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path.
|
||||
|
||||
## [17.2.5] - 2026-08-03
|
||||
|
||||
|
||||
@@ -8,6 +8,14 @@
|
||||
- Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`.
|
||||
- Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`.
|
||||
- Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter.
|
||||
- Replayed xAI encrypted reasoning items on later Responses turns instead of stripping `type: "reasoning"` history.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Invalidated stale paid-xAI model-cache rows written under Chat Completions so the Responses migration takes effect immediately instead of waiting for TTL expiry.
|
||||
- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400.
|
||||
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
|
||||
- Stopped emitting stale `thinking.efforts` dials on paid xAI Responses catalog rows that reject `reasoning.effort` (`grok-code-fast-1`, `grok-build-0.1`, `grok-4.20-0309-reasoning`, and other off-allowlist reasoners).
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
@@ -134,21 +142,6 @@
|
||||
- Fixed dynamic discovery for the `deepseek-v4` model family (such as `deepseek-v4-flash-0731`) under `alibaba-token-plan` missing reasoning configuration and maximum thinking effort.
|
||||
- Fixed GitHub Copilot dynamic discovery retaining stale bundled prices for default-context models instead of using the provider's reported default-tier prices.
|
||||
|
||||
## [17.2.5] - 2026-08-03
|
||||
### Changed
|
||||
|
||||
- Switched the paid xAI provider (`xai` / `XAI_API_KEY`) from Chat Completions to the OpenAI Responses API (`POST https://api.x.ai/v1/responses`), matching SuperGrok `xai-oauth`. Prompt-cache affinity (`x-grok-conv-id`), reasoning-effort allowlisting, and encrypted-reasoning replay rules are now shared across both first-party xAI hosts.
|
||||
- Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`.
|
||||
- Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`.
|
||||
- Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter.
|
||||
- Replayed xAI encrypted reasoning items on later Responses turns instead of stripping `type: "reasoning"` history.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Invalidated stale paid-xAI model-cache rows written under Chat Completions so the Responses migration takes effect immediately instead of waiting for TTL expiry.
|
||||
- Clamped paid xAI Responses `minimal` reasoning effort to `low` (same wire map as SuperGrok) so `xai/grok-4.5` does not 400.
|
||||
- Suppressed presence/frequency penalties and stop sequences on xAI reasoning models so a configured `presencePenalty` does not 400 after the `grok-4.5` default change.
|
||||
|
||||
## [17.2.5] - 2026-08-03
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -22,6 +22,7 @@ import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../sr
|
||||
import {
|
||||
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
|
||||
OPENAI_GPT_56_LONG_CONTEXT_COSTS,
|
||||
applyXaiResponsesThinkingPolicy,
|
||||
resolveWaferServerlessThinkingFormat,
|
||||
} from "../src/provider-models/openai-compat";
|
||||
import type { Api, LongContextTokenCost, Model, ModelSpec } from "../src/types";
|
||||
@@ -353,6 +354,10 @@ export function applyOllamaCloudOutputCap(models: ModelSpec<Api>[]): void {
|
||||
}
|
||||
|
||||
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
if (model.provider === "xai" && model.api === "openai-responses") {
|
||||
const updated = applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">);
|
||||
model.compat = updated.compat;
|
||||
}
|
||||
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
||||
if (copilotLimits) {
|
||||
model.contextWindow = copilotLimits.contextWindow;
|
||||
|
||||
@@ -103730,7 +103730,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-2-1212": {
|
||||
"id": "grok-2-1212",
|
||||
@@ -103749,7 +103756,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-2-latest": {
|
||||
"id": "grok-2-latest",
|
||||
@@ -103768,7 +103782,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-2-vision": {
|
||||
"id": "grok-2-vision",
|
||||
@@ -103788,7 +103809,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 8192,
|
||||
"maxTokens": 4096
|
||||
"maxTokens": 4096,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-2-vision-1212": {
|
||||
"id": "grok-2-vision-1212",
|
||||
@@ -103808,7 +103836,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 8192,
|
||||
"maxTokens": 4096
|
||||
"maxTokens": 4096,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-2-vision-latest": {
|
||||
"id": "grok-2-vision-latest",
|
||||
@@ -103828,7 +103863,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 8192,
|
||||
"maxTokens": 4096
|
||||
"maxTokens": 4096,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-3": {
|
||||
"id": "grok-3",
|
||||
@@ -103847,7 +103889,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-3-fast": {
|
||||
"id": "grok-3-fast",
|
||||
@@ -103866,7 +103915,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-3-fast-latest": {
|
||||
"id": "grok-3-fast-latest",
|
||||
@@ -103885,7 +103941,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-3-latest": {
|
||||
"id": "grok-3-latest",
|
||||
@@ -103904,7 +103967,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 8192
|
||||
"maxTokens": 8192,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-3-mini": {
|
||||
"id": "grok-3-mini",
|
||||
@@ -103930,8 +104000,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-3-mini-fast": {
|
||||
@@ -103958,8 +104039,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-3-mini-fast-latest": {
|
||||
@@ -103986,8 +104078,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-3-mini-latest": {
|
||||
@@ -104014,8 +104117,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-4": {
|
||||
@@ -104036,14 +104150,12 @@
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 64000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4-1-fast": {
|
||||
@@ -104065,14 +104177,12 @@
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4-1-fast-non-reasoning": {
|
||||
@@ -104093,7 +104203,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000
|
||||
"maxTokens": 30000,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4-fast": {
|
||||
"id": "grok-4-fast",
|
||||
@@ -104114,14 +104231,12 @@
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4-fast-non-reasoning": {
|
||||
@@ -104142,7 +104257,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000
|
||||
"maxTokens": 30000,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4.20-0309-non-reasoning": {
|
||||
"id": "grok-4.20-0309-non-reasoning",
|
||||
@@ -104162,7 +104284,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 30000
|
||||
"maxTokens": 30000,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4.20-0309-reasoning": {
|
||||
"id": "grok-4.20-0309-reasoning",
|
||||
@@ -104183,15 +104312,12 @@
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 30000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4.20-beta-latest-non-reasoning": {
|
||||
@@ -104212,7 +104338,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000
|
||||
"maxTokens": 30000,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4.20-beta-latest-reasoning": {
|
||||
"id": "grok-4.20-beta-latest-reasoning",
|
||||
@@ -104233,15 +104366,12 @@
|
||||
},
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 30000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-4.20-multi-agent-beta-latest": {
|
||||
@@ -104269,8 +104399,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-4.3": {
|
||||
@@ -104298,8 +104439,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-4.5": {
|
||||
@@ -104327,8 +104479,19 @@
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "low"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": true,
|
||||
"omitReasoningEffort": false
|
||||
}
|
||||
},
|
||||
"grok-4.6": {
|
||||
@@ -104377,7 +104540,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 4096
|
||||
"maxTokens": 4096,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-build-0.1": {
|
||||
"id": "grok-build-0.1",
|
||||
@@ -104398,14 +104568,12 @@
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 256000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-code-fast-1": {
|
||||
@@ -104426,14 +104594,12 @@
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 10000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
},
|
||||
"grok-vision-beta": {
|
||||
@@ -104454,7 +104620,14 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 8192,
|
||||
"maxTokens": 4096
|
||||
"maxTokens": 4096,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
},
|
||||
"supportsReasoningEffort": false,
|
||||
"omitReasoningEffort": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"xai-oauth": {
|
||||
|
||||
@@ -1376,6 +1376,36 @@ function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): Model
|
||||
// of the omitReasoningEffort gate in pi-ai's stream.ts.
|
||||
const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
|
||||
|
||||
/**
|
||||
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
|
||||
*
|
||||
* models.dev marks many Grok SKUs as reasoners and the thinking rebake would
|
||||
* otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
|
||||
* accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
|
||||
* off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
|
||||
* `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
|
||||
* (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
|
||||
* `xai` rows come from stencil.so and need the same wire facts in the exported
|
||||
* `models.json` so direct catalog readers do not present an unsupported dial.
|
||||
*
|
||||
* Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
|
||||
*/
|
||||
export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
||||
const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
|
||||
return {
|
||||
...model,
|
||||
compat: {
|
||||
...(model.compat ?? {}),
|
||||
reasoningEffortMap: {
|
||||
...XAI_REASONING_EFFORT_MAP,
|
||||
...(model.compat?.reasoningEffortMap ?? {}),
|
||||
},
|
||||
supportsReasoningEffort: effortCapable,
|
||||
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
|
||||
// (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
|
||||
// like it owns `contextWindow`: each entry mirrors its context window. The
|
||||
@@ -5854,7 +5884,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
||||
defaultContextWindow: 131072,
|
||||
}),
|
||||
// --- xAI ---
|
||||
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1"),
|
||||
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
|
||||
transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
|
||||
}),
|
||||
// --- DeepSeek ---
|
||||
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
|
||||
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
|
||||
|
||||
@@ -467,6 +467,49 @@ describe("generated model policies", () => {
|
||||
expect(models[2]?.applyPatchToolType).toBeUndefined();
|
||||
expect(models[3]?.applyPatchToolType).toBeUndefined();
|
||||
});
|
||||
|
||||
it("strips paid xAI Responses effort dials for off-allowlist reasoners", () => {
|
||||
const models: ModelSpec<"openai-responses">[] = [
|
||||
createSpec({
|
||||
id: "grok-code-fast-1",
|
||||
api: "openai-responses",
|
||||
provider: "xai",
|
||||
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
}),
|
||||
createSpec({
|
||||
id: "grok-4.5",
|
||||
api: "openai-responses",
|
||||
provider: "xai",
|
||||
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
}),
|
||||
createSpec({
|
||||
id: "grok-code-fast-1",
|
||||
api: "openai-responses",
|
||||
provider: "openrouter",
|
||||
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.thinking).toBeUndefined();
|
||||
expect(models[0]?.compat).toMatchObject({
|
||||
supportsReasoningEffort: false,
|
||||
omitReasoningEffort: true,
|
||||
reasoningEffortMap: { minimal: "low" },
|
||||
});
|
||||
expect(models[1]?.thinking?.efforts).toEqual([
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
]);
|
||||
expect(models[1]?.compat?.supportsReasoningEffort).toBe(true);
|
||||
// Non-xAI hosts are outside this policy — no baked no-dial compat.
|
||||
expect(models[2]?.thinking).toBeDefined();
|
||||
expect(models[2]?.compat?.supportsReasoningEffort).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("applyOllamaCloudOutputCap", () => {
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import MODELS_JSON from "@oh-my-pi/pi-catalog/models.json" with { type: "json" };
|
||||
import {
|
||||
MODELS_DEV_PROVIDER_DESCRIPTORS,
|
||||
mapModelsDevToModels,
|
||||
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
import { applyGeneratedModelPolicies } from "../scripts/generated-policies";
|
||||
|
||||
const XAI_MODELS_DEV_FIXTURE = {
|
||||
xai: {
|
||||
models: {
|
||||
"grok-4.5": {
|
||||
name: "Grok 4.5",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text", "image"] },
|
||||
limit: { context: 500_000, output: 500_000 },
|
||||
cost: { input: 2, output: 6, cache_read: 0.3 },
|
||||
},
|
||||
"grok-code-fast-1": {
|
||||
name: "Grok Code Fast 1",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text"] },
|
||||
limit: { context: 256_000, output: 10_000 },
|
||||
cost: { input: 0.2, output: 1.5 },
|
||||
},
|
||||
"grok-build-0.1": {
|
||||
name: "Grok Build 0.1",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text", "image"] },
|
||||
limit: { context: 256_000, output: 256_000 },
|
||||
cost: { input: 0, output: 0 },
|
||||
},
|
||||
"grok-4.20-0309-reasoning": {
|
||||
name: "Grok 4.20 (Reasoning)",
|
||||
tool_call: true,
|
||||
reasoning: true,
|
||||
modalities: { input: ["text", "image"] },
|
||||
limit: { context: 2_000_000, output: 64_000 },
|
||||
cost: { input: 2, output: 6 },
|
||||
},
|
||||
"grok-2": {
|
||||
name: "Grok 2",
|
||||
tool_call: true,
|
||||
reasoning: false,
|
||||
modalities: { input: ["text"] },
|
||||
limit: { context: 131_072, output: 8192 },
|
||||
cost: { input: 2, output: 10 },
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
describe("paid xAI Responses thinking policy", () => {
|
||||
it("bakes the effort-dial allowlist on stencil.so → openai-responses mapping", () => {
|
||||
const mapped = mapModelsDevToModels(XAI_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter(
|
||||
model => model.provider === "xai",
|
||||
);
|
||||
const byId = Object.fromEntries(mapped.map(model => [model.id, model]));
|
||||
|
||||
expect(byId["grok-4.5"]?.api).toBe("openai-responses");
|
||||
expect(byId["grok-4.5"]?.compat).toMatchObject({
|
||||
supportsReasoningEffort: true,
|
||||
omitReasoningEffort: false,
|
||||
reasoningEffortMap: { minimal: "low" },
|
||||
});
|
||||
for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) {
|
||||
expect(byId[id]?.reasoning, id).toBe(true);
|
||||
expect(byId[id]?.compat, id).toMatchObject({
|
||||
supportsReasoningEffort: false,
|
||||
omitReasoningEffort: true,
|
||||
reasoningEffortMap: { minimal: "low" },
|
||||
});
|
||||
}
|
||||
expect(byId["grok-2"]?.compat).toMatchObject({
|
||||
supportsReasoningEffort: false,
|
||||
omitReasoningEffort: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("strips stale thinking dials from off-allowlist paid xAI reasoners during generation", () => {
|
||||
const mapped = mapModelsDevToModels(XAI_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter(
|
||||
model => model.provider === "xai",
|
||||
);
|
||||
// Snapshot-era Completions rows still carry a default effort ladder after the
|
||||
// api flip; the generator must not re-emit that dial for Responses.
|
||||
const snapshotStale = mapped.find(model => model.id === "grok-code-fast-1");
|
||||
expect(snapshotStale).toBeDefined();
|
||||
snapshotStale!.thinking = { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
|
||||
|
||||
applyGeneratedModelPolicies(mapped);
|
||||
const byId = Object.fromEntries(mapped.map(model => [model.id, model]));
|
||||
|
||||
expect(byId["grok-4.5"]?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: { minimal: "low" },
|
||||
});
|
||||
for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) {
|
||||
expect(byId[id]?.reasoning, id).toBe(true);
|
||||
expect(byId[id]?.thinking, id).toBeUndefined();
|
||||
expect(byId[id]?.compat, id).toMatchObject({ supportsReasoningEffort: false });
|
||||
}
|
||||
});
|
||||
|
||||
it("exports no-dial rows in the bundled models.json snapshot", () => {
|
||||
const bundled =
|
||||
(MODELS_JSON as unknown as Record<string, Record<string, ModelSpec<"openai-responses">>>).xai ?? {};
|
||||
for (const id of ["grok-code-fast-1", "grok-build-0.1", "grok-4.20-0309-reasoning"] as const) {
|
||||
expect(bundled[id], `xai/${id} missing from models.json`).toBeDefined();
|
||||
expect(bundled[id]?.reasoning, id).toBe(true);
|
||||
expect(bundled[id]?.thinking, id).toBeUndefined();
|
||||
expect(bundled[id]?.compat?.supportsReasoningEffort, id).toBe(false);
|
||||
}
|
||||
expect(bundled["grok-4.5"]?.thinking?.efforts).toContain(Effort.XHigh);
|
||||
expect(bundled["grok-4.5"]?.compat?.supportsReasoningEffort).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -8,6 +8,9 @@
|
||||
- Changed the default model for `XAI_API_KEY` (`xai`) from `grok-4-fast-non-reasoning` to `grok-4.5`.
|
||||
- Changed the default model for SuperGrok OAuth (`xai-oauth`) from `grok-4.3` to `grok-4.5`.
|
||||
- Included `reasoning.encrypted_content` in Responses `include` for paid xAI and SuperGrok OAuth models.
|
||||
- Replayed encrypted xAI reasoning on follow-up Responses turns for `xai` and `xai-oauth`.
|
||||
- Kept automatic model selection on paid `xai/grok-4.5` when only `XAI_API_KEY` is set, instead of preferring SuperGrok `xai-oauth/grok-4.5`.
|
||||
- Stopped sending presence/frequency penalties and stop sequences to xAI reasoning models such as `grok-4.5`, which reject them.
|
||||
|
||||
## [17.3.4] - 2026-08-14
|
||||
|
||||
@@ -364,15 +367,6 @@
|
||||
- Fixed issues with `/btw` branch promotion where branches could park behind active turns, cut from outdated session leaves, or leave rejected branch keys indistinguishable from composer input.
|
||||
- Fixed database bloat by ensuring archived main and nested session rows are properly cleaned up from `stats.db` during garbage collection.
|
||||
- Fixed startup hanging during local model discovery when a timed-out transport left its request pending, which blocked the CLI before OAuth login could finish ([#7482](https://github.com/can1357/oh-my-pi/issues/7482)).
|
||||
### Changed
|
||||
|
||||
- Routed paid xAI models (`XAI_API_KEY` / `xai/…`) through the Responses API used by SuperGrok OAuth instead of Chat Completions.
|
||||
- Changed the default model for `XAI_API_KEY` (`xai`) from `grok-4-fast-non-reasoning` to `grok-4.5`.
|
||||
- Changed the default model for SuperGrok OAuth (`xai-oauth`) from `grok-4.3` to `grok-4.5`.
|
||||
- Included `reasoning.encrypted_content` in Responses `include` for paid xAI and SuperGrok OAuth models.
|
||||
- Replayed encrypted xAI reasoning on follow-up Responses turns for `xai` and `xai-oauth`.
|
||||
- Kept automatic model selection on paid `xai/grok-4.5` when only `XAI_API_KEY` is set, instead of preferring SuperGrok `xai-oauth/grok-4.5`.
|
||||
- Stopped sending presence/frequency penalties and stop sequences to xAI reasoning models such as `grok-4.5`, which reject them.
|
||||
|
||||
## [17.2.5] - 2026-08-03
|
||||
|
||||
|
||||
Reference in New Issue
Block a user