feat(catalog): route paid xAI through Responses like SuperGrok

Switch XAI_API_KEY models from Chat Completions to /v1/responses, default
both xai and xai-oauth to grok-4.5, and include reasoning.encrypted_content.
This commit is contained in:
Yang Yang
2026-08-02 20:29:22 -07:00
parent ffd53ff92a
commit c228dea58b
15 changed files with 288 additions and 116 deletions
+4 -3
View File
@@ -235,12 +235,13 @@ Reasoning fields are not interchangeable.
- Compat needs both policies: disable reasoning for any tool choice, and disable
reasoning only for forced tool choice.
### xAI Grok through Responses/SuperGrok
### xAI Grok through Responses (`xai` and `xai-oauth`)
Keep these independent:
Both the paid API-key provider (`xai` / `XAI_API_KEY`) and SuperGrok OAuth
(`xai-oauth`) chat over `https://api.x.ai/v1/responses`. Keep these independent:
- omit `reasoning.effort`
- include or drop encrypted reasoning replay
- include `reasoning.encrypted_content` (request `include`) vs replay history
- filter reasoning-history wrappers
Some models reject only one of those fields; do not collapse them into one
@@ -5,9 +5,6 @@ import type { AssistantMessage, Context, FetchImpl, Model, SimpleStreamOptions,
import { buildOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1");
if (!model) throw new Error("Expected bundled xAI Grok model");
if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`);
const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] };
const openAI56ResponsesModel = getBundledModel<"openai-responses">("openai", "gpt-5.6");
@@ -44,7 +41,7 @@ function chatCompletionsSse(): Response {
id: "chatcmpl-affinity",
object: "chat.completion.chunk",
created: 0,
model: model.id,
model: openAI56CompletionsModel.id,
choices: [{ index: 0, delta, finish_reason: finishReason }],
});
@@ -56,7 +53,7 @@ function chatCompletionsSse(): Response {
async function captureRequest(
options: OpenAICompletionsOptions,
requestModel: Model<"openai-completions"> = model,
requestModel: Model<"openai-completions"> = openAI56CompletionsModel,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
@@ -83,7 +80,7 @@ async function captureRequest(
async function captureSimpleRequest(
options: SimpleStreamOptions,
requestModel: Model<"openai-completions"> = model,
requestModel: Model<"openai-completions"> = openAI56CompletionsModel,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
@@ -104,51 +101,6 @@ async function captureSimpleRequest(
return { headers: requestHeaders, body };
}
describe("openai-completions xAI cache affinity", () => {
const cases: Array<{
name: string;
options: OpenAICompletionsOptions;
expectedHeader: string | null;
}> = [
{
name: "uses sessionId when no prompt cache key is provided",
options: { sessionId: "session-fallback" },
expectedHeader: "session-fallback",
},
{
name: "keeps the prompt cache key stable across a distinct side-channel session",
options: { promptCacheKey: "stable-cache-key", sessionId: "side-channel-session" },
expectedHeader: "stable-cache-key",
},
{
name: "omits automatic affinity when caching is disabled",
options: {
promptCacheKey: "disabled-cache-key",
sessionId: "disabled-session",
cacheRetention: "none",
},
expectedHeader: null,
},
{
name: "preserves a caller-provided mixed-case affinity header",
options: {
promptCacheKey: "automatic-cache-key",
sessionId: "automatic-session",
headers: { "X-Grok-Conv-Id": "caller-affinity" },
},
expectedHeader: "caller-affinity",
},
];
for (const { name, options, expectedHeader } of cases) {
it(name, async () => {
const { headers } = await captureRequest(options);
expect(headers.get("x-grok-conv-id")).toBe(expectedHeader);
});
}
});
describe("OpenAI Chat Completions explicit prompt cache policy", () => {
const historicalContext: Context = {
messages: [
@@ -57,6 +57,20 @@ const xaiOAuthResponsesModel: Model<"openai-responses"> = {
reasoning: true,
}),
};
const xaiApiKeyResponsesModel: Model<"openai-responses"> = {
...model,
id: "grok-code-fast-1",
name: "Grok Code Fast 1",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
compat: buildOpenAIResponsesCompat({
id: "grok-code-fast-1",
name: "Grok Code Fast 1",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
reasoning: true,
}),
};
const openAI56ResponsesModel: Model<"openai-responses"> = {
...model,
@@ -675,6 +689,16 @@ describe("openai-responses cache affinity", () => {
}
});
it("sets x-grok-conv-id cache affinity for paid xai Responses requests", async () => {
const captured = await captureDispatchedOpenAIResponseHeaders(
{ sessionId: "session-fallback" },
xaiApiKeyResponsesModel,
);
expect(getHeader(captured.headers, "x-grok-conv-id")).toBe("session-fallback");
expect(captured.body?.prompt_cache_key).toBe("session-fallback");
});
it("sets OpenRouter Responses session_id from sessionId in the body", async () => {
const captured = await captureOpenAIResponseHeaders(
{ sessionId: "workflow-123", promptCacheKey: "cache-key-123" },
+1 -1
View File
@@ -1000,7 +1000,7 @@ describe("Generate E2E Tests", () => {
);
});
describe.skipIf(!e2eApiKey("XAI_API_KEY"))("xAI Provider (grok-code-fast-1 via OpenAI Completions)", () => {
describe.skipIf(!e2eApiKey("XAI_API_KEY"))("xAI Provider (grok-code-fast-1 via OpenAI Responses)", () => {
const llm = getBundledModel("xai", "grok-code-fast-1");
it(
@@ -38,6 +38,23 @@ describe("effort-dial-less reasoner encoding (regression)", () => {
expect(grokR.thinking).toBeUndefined();
});
test("paid xai/grok-code-fast-1 reasons but carries no thinking config", () => {
const grokCodeFast = getBundledModel("xai", "grok-code-fast-1");
if (!grokCodeFast) throw new Error("xai/grok-code-fast-1 must be in bundled models.json");
expect(grokCodeFast.api).toBe("openai-responses");
expect(grokCodeFast.reasoning).toBe(true);
expect(grokCodeFast.thinking).toBeUndefined();
expect(getSupportedEfforts(grokCodeFast)).toEqual([]);
});
test("paid xai/grok-4.3 keeps its effort dial", () => {
const grok43 = getBundledModel("xai", "grok-4.3");
if (!grok43) throw new Error("xai/grok-4.3 must be in bundled models.json");
expect(grok43.api).toBe("openai-responses");
expect(grok43.thinking).toBeDefined();
expect(getSupportedEfforts(grok43).length).toBeGreaterThan(0);
});
test("the no-dial encoding stays scoped to openai-responses*", () => {
const claude = getBundledModel("anthropic", "claude-sonnet-4-6");
if (!claude) throw new Error("anthropic/claude-sonnet-4-6 must be in bundled models.json");
@@ -57,6 +74,7 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, undefined, undefined);
expect(params.reasoning).toBeUndefined();
expect(params.include).toContain("reasoning.encrypted_content");
});
test("xai-oauth/grok-4.5 omits unsupported reasoning summary", () => {
@@ -66,5 +84,15 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.reasoning).toEqual({ effort: "high" });
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 requests encrypted reasoning content", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.include).toContain("reasoning.encrypted_content");
});
});
+7
View File
@@ -2,6 +2,13 @@
## [Unreleased]
### Changed
- Switched the paid xAI provider (`xai` / `XAI_API_KEY`) from Chat Completions to the OpenAI Responses API (`POST https://api.x.ai/v1/responses`), matching SuperGrok `xai-oauth`. Prompt-cache affinity (`x-grok-conv-id`), reasoning-effort allowlisting, and encrypted-reasoning replay rules are now shared across both first-party xAI hosts.
- Changed the paid xAI (`XAI_API_KEY`) default model from `grok-4-fast-non-reasoning` to `grok-4.5`.
- Changed the SuperGrok (`xai-oauth`) default model from `grok-4.3` to `grok-4.5`.
- Requested `reasoning.encrypted_content` on first-party xAI Responses calls (`xai` and `xai-oauth`) via the `include` parameter.
## [17.3.4] - 2026-08-14
### Added
+18 -10
View File
@@ -684,24 +684,28 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
const isLocalServingBackend =
(!PROXY_OPENAI_COMPAT_PROVIDERS.has(spec.provider) && LOCAL_OPENAI_COMPAT_PROVIDERS.has(spec.provider)) ||
hasLocalLoopbackBaseUrl(baseUrl);
const isXaiHost = modelMatchesHost({ provider: spec.provider, baseUrl }, "xai");
const compat: ResolvedOpenAIResponsesCompat = {
supportsDeveloperRole: isAzure || isOpenAIUrl || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode: isAzure || detectStrictModeSupport(spec.provider, baseUrl),
supportsReasoningEffort: spec.provider !== "xai-oauth" || isGrokReasoningEffortCapable(id),
// Paid `xai` and SuperGrok `xai-oauth` share api.x.ai `/v1/responses`.
// Only the Grok effort-capable allowlist accepts `reasoning.effort`;
// other reasoners (grok-build, grok-code-fast-1, …) 400 if it is sent.
supportsReasoningEffort: !isXaiHost || isGrokReasoningEffortCapable(id),
supportsLongPromptCacheRetention: isOpenAIUrl,
supportsPromptCacheBreakpoints,
promptCacheBreakpointTtl: supportsPromptCacheBreakpoints ? "30m" : undefined,
// Azure OpenAI and GitHub Copilot Responses paths require tool results
// to strictly match prior tool calls when building Responses inputs.
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
// GitHub Copilot and xAI OAuth reject `detail: "original"` (400 / 422).
// Every other host preserves native-resolution frames (snapcompact relies
// on `original`). Detect Copilot by provider id or base-URL host so a
// model pointed at the Copilot host under a different provider id still
// clamps; xai-oauth is provider-id only (same host family as paid `xai`).
// GitHub Copilot and first-party xAI `/v1/responses` reject
// `detail: "original"` (400 / 422). Every other host preserves
// native-resolution frames (snapcompact relies on `original`). Detect
// Copilot by provider id or base-URL host so a model pointed at the
// Copilot host under a different provider id still clamps.
supportsImageDetailOriginal:
spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
!isXaiHost && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
reasoningEffortMap: {},
supportsReasoningParams: true,
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
@@ -710,8 +714,12 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
includeEncryptedReasoning: spec.provider !== "xai-oauth",
filterReasoningHistory: spec.provider === "xai-oauth" || (isOpenRouter && isAnthropicModel),
// Ask xAI `/v1/responses` for `reasoning.encrypted_content` the same way
// first-party OpenAI Responses does. History still drops `type:
// "reasoning"` wrappers (`filterReasoningHistory`) independently —
// those two flags must not be collapsed.
includeEncryptedReasoning: true,
filterReasoningHistory: isXaiHost || (isOpenRouter && isAnthropicModel),
disableReasoningOnForcedToolChoice: isKimiModel,
disableReasoningOnToolChoice: isDeepseekFamily && reasoningCapable && !isOpenRouter,
supportsToolChoice: true,
@@ -752,7 +760,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.provider) || (id ? MINIMAX_PROVIDER_OR_ID_PATTERN.test(id) : false),
emptyLengthFinishIsContextError: spec.provider === "ollama",
usesOpenAIToolCallIdLimit: spec.provider === "openai",
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
promptCacheSessionHeader: isXaiHost ? "x-grok-conv-id" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : spec.compat?.streamFirstEventTimeoutMs,
streamIdleTimeoutMs: isLocalServingBackend
? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS
+1 -1
View File
@@ -47,7 +47,7 @@ export const KNOWN_HOSTS = {
},
umans: { providers: ["umans"], urlMarkers: ["api.code.umans.ai"] },
xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] },
xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] },
xai: { providers: ["xai", "xai-oauth"], urlMarkers: ["api.x.ai"] },
mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] },
together: { providers: ["together"], urlMarkers: ["api.together.xyz"] },
baseten: { providers: ["baseten"], urlMarkers: ["baseten.co"] },
+39 -39
View File
@@ -103716,7 +103716,7 @@
"grok-2": {
"id": "grok-2",
"name": "Grok 2",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103735,7 +103735,7 @@
"grok-2-1212": {
"id": "grok-2-1212",
"name": "Grok 2 (1212)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103754,7 +103754,7 @@
"grok-2-latest": {
"id": "grok-2-latest",
"name": "Grok 2 Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103773,7 +103773,7 @@
"grok-2-vision": {
"id": "grok-2-vision",
"name": "Grok 2 Vision",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103793,7 +103793,7 @@
"grok-2-vision-1212": {
"id": "grok-2-vision-1212",
"name": "Grok 2 Vision (1212)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103813,7 +103813,7 @@
"grok-2-vision-latest": {
"id": "grok-2-vision-latest",
"name": "Grok 2 Vision Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103833,7 +103833,7 @@
"grok-3": {
"id": "grok-3",
"name": "Grok 3",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103852,7 +103852,7 @@
"grok-3-fast": {
"id": "grok-3-fast",
"name": "Grok 3 Fast",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103871,7 +103871,7 @@
"grok-3-fast-latest": {
"id": "grok-3-fast-latest",
"name": "Grok 3 Fast Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103890,7 +103890,7 @@
"grok-3-latest": {
"id": "grok-3-latest",
"name": "Grok 3 Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -103909,7 +103909,7 @@
"grok-3-mini": {
"id": "grok-3-mini",
"name": "Grok 3 Mini",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -103937,7 +103937,7 @@
"grok-3-mini-fast": {
"id": "grok-3-mini-fast",
"name": "Grok 3 Mini Fast",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -103965,7 +103965,7 @@
"grok-3-mini-fast-latest": {
"id": "grok-3-mini-fast-latest",
"name": "Grok 3 Mini Fast Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -103993,7 +103993,7 @@
"grok-3-mini-latest": {
"id": "grok-3-mini-latest",
"name": "Grok 3 Mini Latest",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104021,7 +104021,7 @@
"grok-4": {
"id": "grok-4",
"name": "Grok 4",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104049,7 +104049,7 @@
"grok-4-1-fast": {
"id": "grok-4-1-fast",
"name": "Grok 4.1 Fast",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104078,7 +104078,7 @@
"grok-4-1-fast-non-reasoning": {
"id": "grok-4-1-fast-non-reasoning",
"name": "Grok 4.1 Fast (Non-Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104098,7 +104098,7 @@
"grok-4-fast": {
"id": "grok-4-fast",
"name": "Grok 4 Fast",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104127,7 +104127,7 @@
"grok-4-fast-non-reasoning": {
"id": "grok-4-fast-non-reasoning",
"name": "Grok 4 Fast (Non-Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104147,7 +104147,7 @@
"grok-4.20-0309-non-reasoning": {
"id": "grok-4.20-0309-non-reasoning",
"name": "Grok 4.20 (Non-Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104167,7 +104167,7 @@
"grok-4.20-0309-reasoning": {
"id": "grok-4.20-0309-reasoning",
"name": "Grok 4.20 (Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104197,7 +104197,7 @@
"grok-4.20-beta-latest-non-reasoning": {
"id": "grok-4.20-beta-latest-non-reasoning",
"name": "Grok 4.20 Beta (Non-Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104217,7 +104217,7 @@
"grok-4.20-beta-latest-reasoning": {
"id": "grok-4.20-beta-latest-reasoning",
"name": "Grok 4.20 Beta (Reasoning)",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104247,7 +104247,7 @@
"grok-4.20-multi-agent-beta-latest": {
"id": "grok-4.20-multi-agent-beta-latest",
"name": "Grok 4.20 Multi-Agent Beta",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104276,7 +104276,7 @@
"grok-4.3": {
"id": "grok-4.3",
"name": "Grok 4.3",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104305,7 +104305,7 @@
"grok-4.5": {
"id": "grok-4.5",
"name": "Grok 4.5",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104363,7 +104363,7 @@
"grok-beta": {
"id": "grok-beta",
"name": "Grok Beta",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104382,7 +104382,7 @@
"grok-build-0.1": {
"id": "grok-build-0.1",
"name": "Grok Build 0.1",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104411,7 +104411,7 @@
"grok-code-fast-1": {
"id": "grok-code-fast-1",
"name": "Grok Code Fast 1",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": true,
@@ -104439,7 +104439,7 @@
"grok-vision-beta": {
"id": "grok-vision-beta",
"name": "Grok Vision Beta",
"api": "openai-completions",
"api": "openai-responses",
"provider": "xai",
"baseUrl": "https://api.x.ai/v1",
"reasoning": false,
@@ -104483,7 +104483,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": true,
@@ -104515,7 +104515,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": true,
@@ -104559,7 +104559,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": false,
@@ -104604,7 +104604,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": false,
@@ -104649,7 +104649,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": false,
@@ -104709,7 +104709,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": true,
@@ -104741,7 +104741,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": true,
@@ -104772,7 +104772,7 @@
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"includeEncryptedReasoning": true,
"filterReasoningHistory": true,
"supportsImageDetailOriginal": false,
"omitReasoningEffort": true,
@@ -111996,4 +111996,4 @@
}
}
}
}
}
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
},
{
id: "xai",
defaultModel: "grok-4-fast-non-reasoning",
defaultModel: "grok-4.5",
envVars: ["XAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
},
{
id: "xai-oauth",
defaultModel: "grok-4.3",
defaultModel: "grok-4.5",
envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
catalogDiscovery: {
@@ -1257,8 +1257,8 @@ export interface XaiModelManagerConfig {
fetch?: FetchImpl;
}
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-completions"> {
return createSimpleOpenAICompletionsOptions("xai", "https://api.x.ai/v1", config);
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
return createSimpleOpenAIResponsesOptions("xai", "https://api.x.ai/v1", config);
}
export interface XaiOAuthModelManagerConfig {
@@ -1352,7 +1352,7 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
const compat = {
...(model.compat ?? {}),
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
@@ -1395,7 +1395,7 @@ function mergeCuratedIntoModel(
const compat = {
...(base.compat ?? {}),
reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
omitReasoningEffort: !effortCapable,
@@ -5713,6 +5713,15 @@ function openAiCompletionsDescriptor(
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
}
function openAiResponsesDescriptor(
modelsDevKey: string,
providerId: string,
baseUrl: string,
options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
): ModelsDevProviderDescriptor {
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
}
function anthropicMessagesDescriptor(
modelsDevKey: string,
providerId: string,
@@ -5837,7 +5846,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
defaultContextWindow: 131072,
}),
// --- xAI ---
openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1"),
// --- DeepSeek ---
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
+29 -4
View File
@@ -223,12 +223,15 @@ describe("buildModel", () => {
});
});
describe("xAI-OAuth Responses reasoning-effort suppression", () => {
const grokResponsesSpec = (id: string): ModelSpec<"openai-responses"> => ({
describe("xAI Responses reasoning-effort suppression", () => {
const grokResponsesSpec = (
id: string,
provider: "xai" | "xai-oauth" = "xai-oauth",
): ModelSpec<"openai-responses"> => ({
id,
name: id,
api: "openai-responses",
provider: "xai-oauth",
provider,
baseUrl: "https://api.x.ai/v1",
reasoning: true,
input: ["text"],
@@ -248,6 +251,28 @@ describe("xAI-OAuth Responses reasoning-effort suppression", () => {
expect(buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3")).supportsReasoningEffort).toBe(true);
});
it("applies the same Responses dialect to paid xai and xai-oauth", () => {
const paid = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3", "xai"));
const oauth = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3", "xai-oauth"));
expect(paid.promptCacheSessionHeader).toBe("x-grok-conv-id");
expect(oauth.promptCacheSessionHeader).toBe("x-grok-conv-id");
expect(paid.includeEncryptedReasoning).toBe(true);
expect(oauth.includeEncryptedReasoning).toBe(true);
expect(paid.filterReasoningHistory).toBe(true);
expect(oauth.filterReasoningHistory).toBe(true);
expect(paid.supportsImageDetailOriginal).toBe(false);
expect(oauth.supportsImageDetailOriginal).toBe(false);
expect(paid.supportsReasoningEffort).toBe(true);
expect(oauth.supportsReasoningEffort).toBe(true);
});
it("omits effort for paid xai models off the Grok allowlist", () => {
const compat = buildOpenAIResponsesCompat(grokResponsesSpec("grok-code-fast-1", "xai"));
expect(compat.supportsReasoningEffort).toBe(false);
expect(compat.omitReasoningEffort).toBe(true);
expect(buildModel(grokResponsesSpec("grok-code-fast-1", "xai")).thinking).toBeUndefined();
});
it("lets an explicit compat.supportsReasoningEffort override the allowlist default", () => {
const compat = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-build"),
@@ -256,7 +281,7 @@ describe("xAI-OAuth Responses reasoning-effort suppression", () => {
expect(compat.supportsReasoningEffort).toBe(true);
});
it("does not suppress effort for a non-xai-oauth provider with a grok-like id", () => {
it("does not suppress effort for a non-xAI provider with a grok-like id", () => {
const compat = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-build"),
provider: "openai",
@@ -0,0 +1,25 @@
import { describe, expect, it } from "bun:test";
import { getBundledModels } from "@oh-my-pi/pi-catalog/models";
import { CATALOG_PROVIDERS } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
import { xaiModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
describe("paid xai (XAI_API_KEY) Responses contract", () => {
it("registers xai on the catalog Responses discovery path", () => {
const entry = CATALOG_PROVIDERS.find(provider => provider.id === "xai");
expect(entry, "xai catalog descriptor").toBeDefined();
expect(entry!.defaultModel).toBe("grok-4.5");
expect(entry!.envVars).toContain("XAI_API_KEY");
const options = xaiModelManagerOptions({ apiKey: "test-key" });
expect(options.providerId).toBe("xai");
expect(options.fetchDynamicModels, "live /v1/models overlay").toBeTypeOf("function");
});
it("bundles every paid xai chat model on openai-responses", () => {
const models = getBundledModels("xai");
expect(models.length).toBeGreaterThan(0);
for (const model of models) {
expect(model.api, `${model.provider}/${model.id}`).toBe("openai-responses");
expect(model.baseUrl).toBe("https://api.x.ai/v1");
}
});
});
@@ -0,0 +1,86 @@
import { describe, expect, it } from "bun:test";
import MODELS_JSON from "@oh-my-pi/pi-catalog/models.json" with { type: "json" };
import { CATALOG_PROVIDERS, DEFAULT_MODEL_PER_PROVIDER } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
import { buildXaiOAuthStaticSeed } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
// Pins the invariant: bundled `models.json` carries every entry the runtime
// curated catalog (XAI_OAUTH_CURATED_MODELS, surfaced via
// buildXaiOAuthStaticSeed) emits. Without this, editing the curated list
// without regenerating `models.json` silently regresses the boot-time
// default-model resolver — the registry sees the runtime seed only after
// `refresh()`, but interactive boot resolves the persisted default
// synchronously from `#loadModels()`, which reads only `models.json`.
//
// Failure here means: run `bun run gen:models` and commit the diff.
describe("xai-oauth bundled catalog (regression)", () => {
const bundled =
(MODELS_JSON as unknown as Record<string, Record<string, ModelSpec<"openai-responses">>>)["xai-oauth"] ?? {};
const seed = buildXaiOAuthStaticSeed();
it("defaults SuperGrok selection to grok-4.5", () => {
const entry = CATALOG_PROVIDERS.find(provider => provider.id === "xai-oauth");
expect(entry?.defaultModel).toBe("grok-4.5");
expect(DEFAULT_MODEL_PER_PROVIDER["xai-oauth"]).toBe("grok-4.5");
expect(bundled["grok-4.5"], "xai-oauth/grok-4.5 must be bundled for the default").toBeDefined();
});
it("bundles every curated id", () => {
const seededIds = seed.map(model => model.id).sort();
const bundledIds = Object.keys(bundled).sort();
expect(bundledIds).toEqual(seededIds);
});
for (const seededModel of seed) {
it(`matches contract for ${seededModel.id}`, () => {
const bundledEntry = bundled[seededModel.id];
expect(bundledEntry, `xai-oauth/${seededModel.id} missing from models.json`).toBeDefined();
expect(bundledEntry.id).toBe(seededModel.id);
expect(bundledEntry.name).toBe(seededModel.name);
expect(bundledEntry.provider).toBe("xai-oauth");
expect(bundledEntry.api).toBe("openai-responses");
expect(bundledEntry.contextWindow).toBe(seededModel.contextWindow);
expect(bundledEntry.reasoning).toBe(seededModel.reasoning);
// Input modality must survive both the curated seed and the bundle.
// Without this the static fallback used on offline boot strips
// vision capability silently (Codex PR #1127 review).
expect(bundledEntry.input).toEqual(seededModel.input);
expect(bundledEntry.compat?.supportsReasoningEffort).toBe(seededModel.compat?.supportsReasoningEffort);
});
}
// Absolute contract for the user-specified SuperGrok addition. The parity
// loop above can't catch a value typo (e.g. 2_000_000) or a flipped
// reasoning flag — both sides regenerate from the same seed together — so
// pin the literal attributes here.
it("exposes grok-composer-2.5-fast as a non-reasoning 200K text model", () => {
const composer = seed.find(model => model.id === "grok-composer-2.5-fast");
expect(composer, "grok-composer-2.5-fast must be in the SuperGrok curated seed").toBeDefined();
expect(composer!.reasoning).toBe(false);
expect(composer!.contextWindow).toBe(200_000);
expect(composer!.input).toEqual(["text"]);
// The bundled models.json entry is byte-identical to the generator's
// deterministic xai-oauth output: gen:models pushes
// buildXaiOAuthStaticSeed() (offline — xai-oauth has no upstream catalog
// source) and applyGeneratedModelPolicies(), so a regen reproduces these
// exact bytes; only unrelated other-provider network churn was excluded
// to keep the diff scoped. Pin its zero-cost invariant (overlay-stable
// for the SuperGrok subscription), which the parity loop above never
// compares. (maxTokens is pinned by the maxTokens-equals-contextWindow
// test below.)
expect(bundled["grok-composer-2.5-fast"]?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
});
// The OAuth surface's /v1/models reports no per-request output limit, so the
// curated catalog owns maxTokens — set to mirror each model's contextWindow
// (the openai-responses wire still clamps the actual request to
// OPENAI_MAX_OUTPUT_TOKENS). Pin maxTokens === contextWindow on both the
// static-seed and bundled paths so a null placeholder can
// never silently leak back into the bundle.
it("sets maxTokens equal to contextWindow for every xai-oauth model", () => {
for (const model of seed) {
expect(model.maxTokens, `seed ${model.id} maxTokens`).toBe(model.contextWindow);
expect(bundled[model.id]?.maxTokens, `bundled ${model.id} maxTokens`).toBe(model.contextWindow);
}
});
});
+7
View File
@@ -2,6 +2,13 @@
## [Unreleased]
### Changed
- Routed paid xAI models (`XAI_API_KEY` / `xai/…`) through the Responses API used by SuperGrok OAuth instead of Chat Completions.
- Changed the default model for `XAI_API_KEY` (`xai`) from `grok-4-fast-non-reasoning` to `grok-4.5`.
- Changed the default model for SuperGrok OAuth (`xai-oauth`) from `grok-4.3` to `grok-4.5`.
- Included `reasoning.encrypted_content` in Responses `include` for paid xAI and SuperGrok OAuth models.
## [17.3.4] - 2026-08-14
### Changed