diff --git a/docs/models.md b/docs/models.md index d1144a222..5ca430d03 100644 --- a/docs/models.md +++ b/docs/models.md @@ -419,7 +419,7 @@ So a model can exist in registry but not be selectable until auth is available. - exact model id (provider inferred) - fuzzy/substring matching - glob scope patterns in `--models` (e.g. `openai/*`, `*sonnet*`) -- optional `:thinkingLevel` suffix (`off|minimal|low|medium|high|xhigh`) +- optional `:thinkingLevel` suffix (`off|minimal|low|medium|high|xhigh|max`) `--provider` is legacy; `--model` is preferred. @@ -582,7 +582,7 @@ Reasoning / thinking: - `supportsReasoningEffort` — accept `reasoning_effort`. Default: auto (off for Grok, Z.ai/Zhipu, and Xiaomi MiMo). - `supportsReasoningParams` — whether request shaping may send reasoning params at all. Default: auto (off for GitHub Copilot chat-completions). -- `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh`) to provider-specific strings (e.g. DeepSeek maps `xhigh -> "max"`). +- `reasoningEffortMap` — partial map from internal effort levels (`minimal|low|medium|high|xhigh|max`) to provider-specific strings (e.g. Fireworks GLM maps `minimal -> "none"`). - `thinkingFormat` — request shape for thinking: `"openai"` (`reasoning_effort`), `"openrouter"` (`reasoning: { effort }`), `"zai"` (`thinking: { type: "enabled" }`), `"qwen"` (top-level `enable_thinking`), or `"qwen-chat-template"` (`chat_template_kwargs.enable_thinking`). Default: `"openai"`. - `reasoningContentField` — assistant field carrying chain-of-thought: `"reasoning_content"`, `"reasoning"`, or `"reasoning_text"`. Default: auto. - `requiresReasoningContentForToolCalls` — assistant tool-call turns must round-trip the reasoning field (DeepSeek-R1, Kimi, OpenRouter when reasoning is on). Default: `false`. diff --git a/docs/rpc.md b/docs/rpc.md index 6e23cf86b..973b80e1b 100644 --- a/docs/rpc.md +++ b/docs/rpc.md @@ -192,7 +192,7 @@ Local-only slash commands may emit `command_output` frames before completing via ```json { "model": { "provider": "...", "id": "..." }, - "thinkingLevel": "off|minimal|low|medium|high|xhigh", + "thinkingLevel": "off|minimal|low|medium|high|xhigh|max", "isStreaming": false, "isCompacting": false, "steeringMode": "all|one-at-a-time", diff --git a/docs/settings.md b/docs/settings.md index 7961c7a1b..5db977693 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -282,7 +282,7 @@ Every key below is defined in the settings schema; `omp config list` shows the f ### Models -`modelRoles`, `modelTags`, and `cycleOrder` work together to define the models you can switch between. Role values may carry a thinking suffix (`:minimal`, `:low`, `:medium`, `:high`, `:xhigh`). +`modelRoles`, `modelTags`, and `cycleOrder` work together to define the models you can switch between. Role values may carry a thinking suffix (`:minimal`, `:low`, `:medium`, `:high`, `:xhigh`, `:max`). ```yaml modelRoles: @@ -342,17 +342,19 @@ thinkingBudgets: medium: 8192 high: 16384 xhigh: 32768 + max: 32768 ``` | Key | Type | Default | Values | |---|---|---|---| -| `defaultThinkingLevel` | enum | `high` | `minimal`, `low`, `medium`, `high`, `xhigh`, `auto`. Override per run with `--thinking`. | +| `defaultThinkingLevel` | enum | `high` | `minimal`, `low`, `medium`, `high`, `xhigh`, `max`, `auto`. Override per run with `--thinking`. | | `hideThinkingBlock` | boolean | `false` | Hide thinking blocks in output. `--hide-thinking` sets it for the run (display only). | | `thinkingBudgets.minimal` | number | `1024` | Token budget for the `minimal` level. | | `thinkingBudgets.low` | number | `2048` | Token budget for `low`. | | `thinkingBudgets.medium` | number | `8192` | Token budget for `medium`. | | `thinkingBudgets.high` | number | `16384` | Token budget for `high`. | | `thinkingBudgets.xhigh` | number | `32768` | Token budget for `xhigh`. | +| `thinkingBudgets.max` | number | `32768` | Token budget for `max`. | ### Sampling diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 109f6905a..c8702a17d 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `ThinkingLevel.Max` ("max") above `xhigh`, mapping to the catalog `Effort.Max` tier. + ### Fixed - Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`. diff --git a/packages/agent/README.md b/packages/agent/README.md index 87c3b2435..148b7182f 100644 --- a/packages/agent/README.md +++ b/packages/agent/README.md @@ -134,7 +134,7 @@ const agent = new Agent({ initialState: { systemPrompt: string[], model: Model, - thinkingLevel: "off" | "minimal" | "low" | "medium" | "high" | "xhigh", + thinkingLevel: "off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max", tools: AgentTool[], messages: AgentMessage[], }, diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 772f0b232..8acf7ffb8 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -659,6 +659,8 @@ function effortFromThinkingLevel(level: ThinkingLevel): Effort { return Effort.High; case ThinkingLevel.XHigh: return Effort.XHigh; + case ThinkingLevel.Max: + return Effort.Max; case ThinkingLevel.Off: case ThinkingLevel.Inherit: throw new Error(`effortFromThinkingLevel: ${level} must be handled by caller`); diff --git a/packages/agent/src/thinking.ts b/packages/agent/src/thinking.ts index e89c1e834..3dcd6611f 100644 --- a/packages/agent/src/thinking.ts +++ b/packages/agent/src/thinking.ts @@ -13,6 +13,7 @@ export const ThinkingLevel = { Medium: Effort.Medium, High: Effort.High, XHigh: Effort.XHigh, + Max: Effort.Max, } as const; export type ThinkingLevel = (typeof ThinkingLevel)[keyof typeof ThinkingLevel]; diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7f0c5dc5f..b2c3669e1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,10 +4,14 @@ ### Added +- Added `max` as a first-class reasoning effort option across providers and wire schemas +- Updated `max` reasoning budget to 32768 tokens across all provider budget tables + - Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. - Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. - Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. - Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). +- Added `"max"` as a first-class reasoning effort across provider options, wire types (`reasoning_effort`/`reasoning.effort`), server intake guards, and the Codex request transformer; user `max` now serializes 1:1 to the provider `max` tier instead of being reachable only through the retired shifted effort maps. Inbound Anthropic gateway requests now map `output_config.effort` onto `options.reasoning`. ### Changed @@ -16,6 +20,7 @@ - Standardized Responses Lite activation via model-level catalog flags - Recognized Pro Lite as a paid plan tier for OpenAI Codex models - Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. +- Changed effort budget tables (`ANTHROPIC_THINKING`, `GOOGLE_THINKING`, `BEDROCK_CLAUDE_THINKING`, Bedrock `defaultBudgets`) to carry a `max` row (32768), and `getGoogleBudget` to resolve `max` to the largest bucket explicitly. ### Fixed diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index f361692bc..b30d8a9aa 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -997,6 +997,7 @@ function buildAdditionalModelRequestFields( medium: 8192, high: 16384, xhigh: 32768, + max: 32768, }; const budget = options.thinkingBudgets?.[level] ?? defaultBudgets[level]; diff --git a/packages/ai/src/providers/anthropic-messages-server.ts b/packages/ai/src/providers/anthropic-messages-server.ts index 480d00931..63616473a 100644 --- a/packages/ai/src/providers/anthropic-messages-server.ts +++ b/packages/ai/src/providers/anthropic-messages-server.ts @@ -1,3 +1,4 @@ +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { logger } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import { captureRequestHeaders, resolvePromptCacheKey } from "../auth-gateway/http"; @@ -291,6 +292,19 @@ function deriveCacheRetention(data: { return strongest; } +/** + * Inbound `output_config.effort` wire literal → catalog `Effort` (1:1). + * Values outside this table (none exist in the schema today) are ignored + * rather than guessed at. + */ +const REASONING_EFFORT_BY_WIRE: Partial> = { + low: Effort.Low, + medium: Effort.Medium, + high: Effort.High, + xhigh: Effort.XHigh, + max: Effort.Max, +}; + export function parseRequest(body: unknown, headers?: Headers): ParsedRequest { const data = anthropicMessagesRequestSchema(body); if (data instanceof type.errors) { @@ -351,6 +365,10 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest { if (data.output_config?.task_budget) { options.taskBudget = data.output_config.task_budget; } + if (data.output_config?.effort) { + const mapped = REASONING_EFFORT_BY_WIRE[data.output_config.effort]; + if (mapped !== undefined) options.reasoning = mapped; + } const cacheRetention = deriveCacheRetention(data); if (cacheRetention !== undefined) options.cacheRetention = cacheRetention; // Anthropic clients commonly send `metadata: { user_id }`; forward verbatim diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 31090702c..2c8e73430 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -56,7 +56,7 @@ function resolveDeploymentName(model: Model<"azure-openai-responses">, options?: // Azure OpenAI Responses-specific options export interface AzureOpenAIResponsesOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "detailed" | "concise" | null; azureApiVersion?: string; azureResourceName?: string; diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index a0fb862a4..bfcdb65e0 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -35,7 +35,7 @@ import { transformMessages } from "./transform-messages"; import { joinTextWithImagePlaceholder, partitionVisionContent } from "./vision-guard"; export interface OllamaChatOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; disableReasoning?: boolean; toolChoice?: ToolChoice; } diff --git a/packages/ai/src/providers/openai-chat-server-schema.ts b/packages/ai/src/providers/openai-chat-server-schema.ts index bdb1835e8..2490eec53 100644 --- a/packages/ai/src/providers/openai-chat-server-schema.ts +++ b/packages/ai/src/providers/openai-chat-server-schema.ts @@ -209,7 +209,7 @@ export const openaiChatRequestSchema = type({ "frequency_penalty?": "number", "logit_bias?": type({ "[string]": "number" }), "user?": "string", - "reasoning_effort?": "'minimal' | 'low' | 'medium' | 'high' | 'xhigh'", + "reasoning_effort?": "'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'", "parallel_tool_calls?": "boolean", "service_tier?": "'auto' | 'default' | 'flex' | 'scale' | 'priority'", "metadata?": type({ "[string]": "unknown" }), diff --git a/packages/ai/src/providers/openai-chat-server.ts b/packages/ai/src/providers/openai-chat-server.ts index 441138ea9..d2f8e0bbc 100644 --- a/packages/ai/src/providers/openai-chat-server.ts +++ b/packages/ai/src/providers/openai-chat-server.ts @@ -35,7 +35,14 @@ export type { ParsedRequest }; type ReasoningEffort = NonNullable; function isReasoningEffort(value: unknown): value is ReasoningEffort { - return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; + return ( + value === "minimal" || + value === "low" || + value === "medium" || + value === "high" || + value === "xhigh" || + value === "max" + ); } function isServiceTier(value: unknown): value is ServiceTier { diff --git a/packages/ai/src/providers/openai-chat-wire.ts b/packages/ai/src/providers/openai-chat-wire.ts index b2e5f22a0..0ff7f4a9c 100644 --- a/packages/ai/src/providers/openai-chat-wire.ts +++ b/packages/ai/src/providers/openai-chat-wire.ts @@ -116,7 +116,7 @@ export type Metadata = { }; /** Constrains effort on reasoning for reasoning models. */ -export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null; +export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | null; /** JSON object response format (older JSON mode). */ export interface ResponseFormatJSONObject { diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 609a570fb..a96d5b2a9 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -110,7 +110,7 @@ import { import { transformMessages } from "./transform-messages"; export interface OpenAICodexResponsesOptions extends StreamOptions { - reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "concise" | "detailed" | null; /** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 8c095137f..a25d5b665 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -8,7 +8,7 @@ import { mapOpenAIReasoningEffort } from "../openai-shared"; export type CodexReasoningContext = "auto" | "current_turn" | "all_turns"; /** User-facing effort levels accepted by Codex request options. */ -type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; +type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; /** Caller literal → catalog `Effort` bridge (the enum is nominal). */ const EFFORT_BY_NAME: Record = { @@ -17,6 +17,7 @@ const EFFORT_BY_NAME: Record = { medium: Effort.Medium, high: Effort.High, xhigh: Effort.XHigh, + max: Effort.Max, }; export interface ReasoningConfig { @@ -28,7 +29,7 @@ export interface ReasoningConfig { } export interface CodexRequestOptions { - /** User-facing effort; the wire-only `max` tier is reached via the model's effort map. */ + /** User-facing effort; maps 1:1 onto the wire tier of the same name. */ reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ @@ -99,7 +100,8 @@ export function resolveCodexResponsesLite( /** * Clamp a user-facing effort to the model's ladder, then remap to the wire - * tier (e.g. GPT-5.6's shifted five-tier scale sends `max` for user `xhigh`). + * tier. User efforts map 1:1 onto wire tiers; the effort map only covers + * host quirks where a wire tier genuinely does not exist (e.g. `minimal→none`). * A mapped value outside the Codex wire vocabulary is a broken compat/model * effort map — fail loudly rather than silently sending a different tier. */ diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index f200c268c..83cc11666 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -459,7 +459,7 @@ export function isOpenAICompletionsProgressChunk(chunk: unknown): boolean { export interface OpenAICompletionsOptions extends StreamOptions { toolChoice?: ToolChoice; - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; /** Force-disable reasoning where supported, or request the lowest effort on generic effort endpoints. */ disableReasoning?: boolean; serviceTier?: ServiceTier; diff --git a/packages/ai/src/providers/openai-responses-server.ts b/packages/ai/src/providers/openai-responses-server.ts index 6a4d947d8..e2dd73389 100644 --- a/packages/ai/src/providers/openai-responses-server.ts +++ b/packages/ai/src/providers/openai-responses-server.ts @@ -40,7 +40,14 @@ export type { ParsedRequest }; // ─── narrow guards ────────────────────────────────────────────────────────── function isReasoningEffort(value: unknown): value is NonNullable { - return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; + return ( + value === "minimal" || + value === "low" || + value === "medium" || + value === "high" || + value === "xhigh" || + value === "max" + ); } function isServiceTier(value: unknown): value is NonNullable { diff --git a/packages/ai/src/providers/openai-responses-wire.ts b/packages/ai/src/providers/openai-responses-wire.ts index 7992ec319..e194a4c3f 100644 --- a/packages/ai/src/providers/openai-responses-wire.ts +++ b/packages/ai/src/providers/openai-responses-wire.ts @@ -6305,9 +6305,9 @@ export interface Reasoning { /** * Constrains effort on reasoning for * [reasoning models](https://platform.openai.com/docs/guides/reasoning). Currently - * supported values are `none`, `minimal`, `low`, `medium`, `high`, and `xhigh`. - * Reducing reasoning effort can result in faster responses and fewer tokens used - * on reasoning in a response. + * supported values are `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, and + * `max`. Reducing reasoning effort can result in faster responses and fewer + * tokens used on reasoning in a response. * * - `gpt-5.1` defaults to `none`, which does not perform reasoning. The supported * reasoning values for `gpt-5.1` are `none`, `low`, `medium`, and `high`. Tool @@ -6316,6 +6316,7 @@ export interface Reasoning { * support `none`. * - The `gpt-5-pro` model defaults to (and only supports) `high` reasoning effort. * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. + * - `max` is supported for `gpt-5.6` and later models. */ effort?: ReasoningEffort | null; /** @@ -6346,9 +6347,9 @@ export interface Reasoning { /** * Constrains effort on reasoning for * [reasoning models](https://platform.openai.com/docs/guides/reasoning). Currently - * supported values are `none`, `minimal`, `low`, `medium`, `high`, and `xhigh`. - * Reducing reasoning effort can result in faster responses and fewer tokens used - * on reasoning in a response. + * supported values are `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, and + * `max`. Reducing reasoning effort can result in faster responses and fewer tokens + * used on reasoning in a response. * * - `gpt-5.1` defaults to `none`, which does not perform reasoning. The supported * reasoning values for `gpt-5.1` are `none`, `low`, `medium`, and `high`. Tool @@ -6357,8 +6358,9 @@ export interface Reasoning { * support `none`. * - The `gpt-5-pro` model defaults to (and only supports) `high` reasoning effort. * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. + * - `max` is supported for `gpt-5.6` and later models. */ -export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null; +export type ReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | null; /** * JSON object response format. An older method of generating JSON responses. Using * `json_schema` is recommended for models that support it. Note that the model diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 937cbe87a..2be20dd81 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -95,7 +95,7 @@ import { // OpenAI Responses-specific options export interface OpenAIResponsesOptions extends StreamOptions { - reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh"; + reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "detailed" | "concise" | null; serviceTier?: ServiceTier; textVerbosity?: "low" | "medium" | "high"; diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 789f31213..47a803349 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -643,7 +643,7 @@ export type OpenAICompletionsParams = Omit = { medium: 8192, high: 16384, xhigh: 32768, + max: 32768, }; const GOOGLE_THINKING: Record = { @@ -1252,6 +1253,7 @@ const GOOGLE_THINKING: Record = { medium: 8192, high: 16384, xhigh: 24575, + max: 32768, }; const BEDROCK_CLAUDE_THINKING: Record = { @@ -1260,6 +1262,7 @@ const BEDROCK_CLAUDE_THINKING: Record = { medium: 8192, high: 16384, xhigh: 16384, + max: 32768, }; function resolveBedrockThinkingBudget( @@ -1842,7 +1845,9 @@ function getGoogleBudget( return 2048; case "medium": return 8192; - default: + case "high": + case "xhigh": + case "max": return model.id.includes("2.5-flash") ? 24576 : 32768; } } diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index d3e7b42f5..837bf7b21 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -1939,16 +1939,17 @@ describe("Anthropic request fingerprint alignment", () => { }); it("drops sampling params and keeps summarized adaptive thinking for OAuth Opus 4.7+", async () => { + const opus47 = buildModel({ + ...ANTHROPIC_MODEL_SPEC, + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + }, + }); const payload = (await captureAnthropicPayload( - buildModel({ - ...ANTHROPIC_MODEL_SPEC, - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - thinking: { - mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - }, - }), + opus47, { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1976,18 +1977,10 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.context_management).toEqual({ edits: [{ type: "clear_thinking_20251015", keep: "all" }], }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); - const maxPayload = (await captureAnthropicPayload( - buildModel({ - ...ANTHROPIC_MODEL_SPEC, - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - thinking: { - mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - }, - }), + const xhighPayload = (await captureAnthropicPayload( + opus47, { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -2000,6 +1993,23 @@ describe("Anthropic request fingerprint alignment", () => { thinking?: { type?: string; display?: string }; output_config?: { effort?: string }; }; + expect(xhighPayload.thinking).toEqual({ type: "adaptive", display: "summarized" }); + expect(xhighPayload.output_config).toEqual({ effort: "xhigh" }); + + const maxPayload = (await captureAnthropicPayload( + opus47, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + }, + { + thinkingEnabled: true, + reasoning: Effort.Max, + }, + )) as { + thinking?: { type?: string; display?: string }; + output_config?: { effort?: string }; + }; expect(maxPayload.thinking).toEqual({ type: "adaptive", display: "summarized" }); expect(maxPayload.output_config).toEqual({ effort: "max" }); }); @@ -2014,15 +2024,14 @@ describe("Anthropic request fingerprint alignment", () => { baseUrl: "https://api.code.umans.ai", thinking: { mode: "anthropic-budget-effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }, }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], }, - Effort.XHigh, + Effort.Max, )) as { thinking?: { type?: string; budget_tokens?: number }; output_config?: { effort?: string }; @@ -2098,7 +2107,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2120,7 +2129,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.context_management).toEqual({ edits: [{ type: "clear_thinking_20251015", keep: "all" }], }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); }); it("sends task budgets through Anthropic output_config without dropping adaptive effort", async () => { @@ -2131,7 +2140,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2151,7 +2160,7 @@ describe("Anthropic request fingerprint alignment", () => { }; expect(payload.output_config).toEqual({ - effort: "xhigh", + effort: "high", task_budget: { type: "tokens", total: 64_000, remaining: 48_000 }, }); }); @@ -2164,7 +2173,7 @@ describe("Anthropic request fingerprint alignment", () => { name: "Claude Opus 4.7", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2209,7 +2218,7 @@ describe("Anthropic request fingerprint alignment", () => { maxTokens: 128_000, thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, }), { @@ -2236,7 +2245,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.tool_choice).toEqual({ type: "auto" }); expect(payload.thinking).toEqual({ type: "adaptive", display: "summarized" }); - expect(payload.output_config).toEqual({ effort: "xhigh" }); + expect(payload.output_config).toEqual({ effort: "high" }); } }); diff --git a/packages/ai/test/auth-gateway-anthropic-messages.test.ts b/packages/ai/test/auth-gateway-anthropic-messages.test.ts index dce464548..1df22f39e 100644 --- a/packages/ai/test/auth-gateway-anthropic-messages.test.ts +++ b/packages/ai/test/auth-gateway-anthropic-messages.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { encodeResponse, encodeStream, parseRequest } from "@oh-my-pi/pi-ai/providers/anthropic-messages-server"; import type { AssistantMessage, AssistantMessageEvent, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; function emptyUsage(): AssistantMessage["usage"] { return { @@ -220,6 +221,32 @@ describe("anthropic-messages parseRequest", () => { expect(parsed.context.messages[1]!.role).toBe("toolResult"); }); + it("maps inbound output_config.effort onto options.reasoning 1:1", () => { + const cases = [ + ["low", Effort.Low], + ["medium", Effort.Medium], + ["high", Effort.High], + ["xhigh", Effort.XHigh], + ["max", Effort.Max], + ] as const; + for (const [wire, effort] of cases) { + const parsed = parseRequest({ + model: "m", + max_tokens: 8, + output_config: { effort: wire }, + messages: [{ role: "user", content: "hi" }], + }); + expect(parsed.options.reasoning).toBe(effort); + } + + const absent = parseRequest({ + model: "m", + max_tokens: 8, + messages: [{ role: "user", content: "hi" }], + }); + expect(absent.options.reasoning).toBeUndefined(); + }); + it("rejects missing required fields and unsupported request controls", () => { expect(() => parseRequest({})).toThrow(/model/); expect(() => parseRequest({ model: "m", messages: [] })).toThrow(/max_tokens/); diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index 3cbdfd349..f651f5714 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -3,6 +3,7 @@ import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; interface OpenAICompletionAssistantWireMessage { @@ -66,52 +67,37 @@ function assistantToolCall( describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- - // Fix 1: effortMap for DeepSeek-family on any provider + // Fix 1: honest [high, max] ladder for DeepSeek-family on any provider // ---------------------------------------------------------------- - describe("thinking effortMap (Fix 1)", () => { - it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => { + describe("thinking ladder (Fix 1)", () => { + it("bakes the honest [high, max] ladder with no effortMap on opencode-go", () => { const model = deepseekModel({ provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); - it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => { + it("bakes the honest [high, max] ladder with no effortMap on NVIDIA", () => { const model = deepseekModel({ provider: "nvidia", baseUrl: "https://integrate.api.nvidia.com/v1", id: "deepseek-ai/deepseek-v4-flash", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); - it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => { + it("bakes the honest [high, max] ladder with no effortMap on the official endpoint", () => { const model = deepseekModel({ provider: "deepseek", baseUrl: "https://api.deepseek.com/v1", id: "deepseek-v4-pro", }); - expect(model.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); it("does NOT map xhigh for non-DeepSeek models", () => { diff --git a/packages/ai/test/glm-5.2-reasoning-effort.test.ts b/packages/ai/test/glm-5.2-reasoning-effort.test.ts index 969e4291f..1cb4246f0 100644 --- a/packages/ai/test/glm-5.2-reasoning-effort.test.ts +++ b/packages/ai/test/glm-5.2-reasoning-effort.test.ts @@ -6,11 +6,12 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; // GLM-5.2 reasoning-effort dialects diverge per host (verified against live -// endpoints): a direct GLM host (Fireworks) wants the top UI tier on the wire -// as `max` while keeping its distinct lower tiers, whereas OpenRouter rejects -// `max` (HTTP 400) and treats `xhigh` as its own max tier. The catalog bakes -// the right `thinking.effortMap`; these tests pin the resulting wire value so a -// future map change can't silently 400 either host. +// endpoints): a direct GLM host (Fireworks) exposes a real `max` top tier and +// keeps its distinct lower tiers (with the `minimal -> none` host quirk), +// whereas OpenRouter rejects `max` (HTTP 400) and treats `xhigh` as its own +// max tier. The catalog bakes the right ladder/`thinking.effortMap`; these +// tests pin the resulting wire value so a future change can't silently 400 +// either host. const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }] }; function chatSse(): Response { @@ -105,8 +106,8 @@ const openRouter = buildModel({ describe("GLM-5.2 reasoning effort wire mapping", () => { afterEach(() => vi.restoreAllMocks()); - it("maps the top tier to reasoning_effort:max on a direct GLM host (Fireworks), lower tiers literal", async () => { - expect(await captureChatEffort(fireworks, Effort.XHigh)).toBe("max"); + it("sends reasoning_effort:max for the real max tier on a direct GLM host (Fireworks), lower tiers literal", async () => { + expect(await captureChatEffort(fireworks, Effort.Max)).toBe("max"); expect(await captureChatEffort(fireworks, Effort.High)).toBe("high"); expect(await captureChatEffort(fireworks, Effort.Medium)).toBe("medium"); // Fireworks rejects literal `minimal`; the host quirk merge keeps `minimal -> none`. diff --git a/packages/ai/test/max-effort-wire.test.ts b/packages/ai/test/max-effort-wire.test.ts new file mode 100644 index 000000000..d716fe221 --- /dev/null +++ b/packages/ai/test/max-effort-wire.test.ts @@ -0,0 +1,158 @@ +import { describe, expect, it, vi } from "bun:test"; +import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import { transformRequestBody } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { createCodexModel } from "./helpers"; + +// End-to-end guard for the first-class `max` reasoning tier: a user-requested +// `reasoning: "max"` on a model whose ladder natively includes `Effort.Max` +// must reach every wire surface verbatim — no aliasing, no clamping. Fixtures +// use explicit thinking ladders and neutral ids so catalog detection cannot +// interfere. + +const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }] }; + +const MAX_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max] as const; + +function chatSse(): Response { + const chunk = (delta: unknown, finish: string | null) => + JSON.stringify({ + id: "x", + object: "chat.completion.chunk", + created: 0, + choices: [{ index: 0, delta, finish_reason: finish }], + }); + return new Response(`data: ${chunk({ content: "ok" }, null)}\n\ndata: ${chunk({}, "stop")}\n\ndata: [DONE]\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function responsesSse(): Response { + return new Response( + `data: ${JSON.stringify({ + type: "response.completed", + response: { + status: "completed", + usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } }, + }, + })}\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +describe("first-class max reasoning tier wire coverage", () => { + it("sends reasoning_effort:max on Chat Completions", async () => { + const model: Model<"openai-completions"> = buildModel({ + id: "max-wire-chat", + name: "Max Wire Chat", + api: "openai-completions", + provider: "custom", + baseUrl: "https://chat.example.test/v1", + reasoning: true, + compat: { + thinkingFormat: "openai", + supportsReasoningParams: true, + supportsReasoningEffort: true, + }, + thinking: { mode: "effort", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_384, + }); + + let body: Record | undefined; + const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + body = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return chatSse(); + }); + for await (const event of streamOpenAICompletions(model, context, { + apiKey: "k", + fetch: fetchMock, + reasoning: "max", + })) { + if (event.type === "done" || event.type === "error") break; + } + expect(body?.reasoning_effort).toBe("max"); + }); + + it("sends reasoning.effort:max on the Responses surface", async () => { + const model: Model<"openai-responses"> = buildModel({ + id: "max-wire-responses", + name: "Max Wire Responses", + api: "openai-responses", + provider: "custom-responses", + baseUrl: "https://responses.example.test/v1", + reasoning: true, + compat: { + supportsReasoningParams: true, + supportsReasoningEffort: true, + }, + thinking: { mode: "effort", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_384, + }); + + let body: Record | undefined; + const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + body = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return responsesSse(); + }); + for await (const event of streamOpenAIResponses(model, context, { + apiKey: "k", + fetch: fetchMock, + reasoning: "max", + })) { + if (event.type === "done" || event.type === "error") break; + } + const reasoningParam = body?.reasoning as { effort?: string } | undefined; + expect(reasoningParam?.effort).toBe("max"); + }); + + it("sends reasoning.effort:max through the Codex request transformer", async () => { + const model = createCodexModel("codex-max-wire", { + thinking: { mode: "effort", efforts: MAX_LADDER }, + }); + const transformed = await transformRequestBody({ model: model.id, input: [] }, model, { + reasoningEffort: "max", + }); + expect(transformed.reasoning?.effort).toBe("max"); + }); + + it("sends output_config.effort:max on Anthropic adaptive thinking", async () => { + const model = buildModel({ + id: "adaptive-max-wire", + name: "Adaptive Max Wire", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + thinking: { mode: "anthropic-adaptive", efforts: MAX_LADDER }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }); + + const { promise, resolve } = Promise.withResolvers(); + const controller = new AbortController(); + controller.abort(); + streamAnthropic(model, context, { + apiKey: "sk-ant-test", + isOAuth: false, + signal: controller.signal, + thinkingEnabled: true, + reasoning: Effort.Max, + onPayload: payload => resolve(payload), + }); + const payload = (await promise) as { output_config?: { effort?: string } }; + expect(payload.output_config).toEqual({ effort: "max" }); + }); +}); diff --git a/packages/ai/test/ollama-reasoning-effort-backfill.test.ts b/packages/ai/test/ollama-reasoning-effort-backfill.test.ts index 8cafe1f2a..04d0ea2af 100644 --- a/packages/ai/test/ollama-reasoning-effort-backfill.test.ts +++ b/packages/ai/test/ollama-reasoning-effort-backfill.test.ts @@ -3,6 +3,7 @@ import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-response import type { Context } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { clampThinkingLevelForModel, getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; const testContext: Context = { messages: [{ role: "user", content: "hi", timestamp: 0 }], @@ -14,11 +15,13 @@ function abortedSignal(): AbortSignal { return controller.signal; } -describe("ollama reasoning effort backfill reaches the Responses wire", () => { - it("sends low instead of minimal for a stale ollama spec carrying no effort map", async () => { - // Reproduces the HTTP 400 `invalid reasoning value: "minimal"` path: a - // reasoning-capable Ollama model whose cached/custom spec predates the - // remap. buildModel must backfill the effort map so the wire sends `low`. +describe("ollama effort ladder normalization reaches the Responses wire", () => { + it("normalizes a stale ollama spec and sends native max on the wire", async () => { + // A cached/custom spec from before the wire-exact ladder existed: + // reasoning-capable with `minimal` offered. buildModel must normalize + // the ladder to Ollama's low/medium/high/max vocabulary so requests at + // the top tier serialize `max` verbatim (HTTP 400 `invalid reasoning + // value: "minimal"` was the historical failure of the stale surface). const model = buildModel({ id: "gemma4:e4b", name: "gemma4:e4b", @@ -33,16 +36,20 @@ describe("ollama reasoning effort backfill reaches the Responses wire", () => { thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, }); + // The stale `minimal` tier is gone; selecting it clamps to the floor. + expect(getSupportedEfforts(model)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Low); + const { promise, resolve } = Promise.withResolvers>(); streamOpenAIResponses(model, testContext, { apiKey: "test-key", signal: abortedSignal(), - reasoning: "minimal", + reasoning: "max", reasoningSummary: "auto", onPayload: payload => resolve(payload as Record), }); const payload = await promise; - expect(payload.reasoning).toEqual({ effort: "low", summary: "auto" }); + expect(payload.reasoning).toEqual({ effort: "max", summary: "auto" }); }); }); diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index 1c7492afc..d174d7071 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -312,28 +312,29 @@ describe("openai-codex reasoning effort validation", () => { transformRequestBody({ ...body }, createCodexModel(body.model), { reasoningEffort: "xhigh" }), ).rejects.toThrow(/Supported efforts: medium, high/); }); + + it("rejects gpt-5.6 minimal now that the wire floor is low", async () => { + const body: RequestBody = { model: "gpt-5.6-sol", input: [] }; + await expect( + transformRequestBody(body, createCodexModel(body.model), { reasoningEffort: "minimal" }), + ).rejects.toThrow(/Supported efforts: low, medium, high, xhigh, max/); + }); }); describe("openai-codex reasoning effort wire mapping", () => { - it("shifts gpt-5.6 user efforts one wire tier up via the baked effort map", async () => { + it("maps gpt-5.6 user efforts 1:1 onto wire tiers", async () => { const model = createCodexModel("gpt-5.6-sol"); - const shifted = [ - ["minimal", "low"], - ["low", "medium"], - ["medium", "high"], - ["high", "xhigh"], - ["xhigh", "max"], - ] as const; + const efforts = ["low", "medium", "high", "xhigh", "max"] as const; - for (const [requested, wire] of shifted) { + for (const effort of efforts) { const transformed = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: requested, + reasoningEffort: effort, }); - expect(transformed.reasoning?.effort).toBe(wire); + expect(transformed.reasoning?.effort).toBe(effort); } }); - it("keeps pre-5.6 efforts unshifted and passes none through unmapped", async () => { + it("keeps pre-5.6 efforts 1:1 and passes none through unmapped", async () => { const gpt55 = createCodexModel("gpt-5.5"); const unshifted = await transformRequestBody({ model: gpt55.id }, gpt55, { reasoningEffort: "xhigh" }); expect(unshifted.reasoning?.effort).toBe("xhigh"); diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 728124025..4c9d8ff0e 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -115,7 +115,7 @@ function kimiZaiModel(): Model<"openai-completions"> { async function captureOpenAICompletionsPayload( model: Model<"openai-completions">, context: Context = baseContext(), - options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" }, + options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max" }, ): Promise { const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); @@ -810,7 +810,7 @@ describe("openai-completions compatibility", () => { expect(getNestedBoolean(chatTemplateArgs, "enable_thinking")).toBe(true); }); - it("maps GLM-5.2 xhigh to Z.AI max and enables tool streaming", async () => { + it("sends reasoning_effort:max for the real Z.AI max tier and enables tool streaming", async () => { const model = zaiGlm52Model(); const readTool: Tool = { name: "read", @@ -828,7 +828,7 @@ describe("openai-completions compatibility", () => { { ...baseContext(), tools: [readTool] }, { apiKey: "test-key", - reasoning: "xhigh", + reasoning: "max", signal: createAbortedSignal(), onPayload: payload => resolve(payload), maxTokens: 65_536, @@ -879,21 +879,10 @@ describe("openai-completions compatibility", () => { expect(payloadObject?.tool_stream).toBeUndefined(); }); - it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => { + it("bakes the honest [high, max] Z.AI GLM-5.2 ladder with no effortMap", () => { const model = zaiGlm52Model(); - - const { promise, resolve } = Promise.withResolvers(); - streamOpenAICompletions(model, baseContext(), { - apiKey: "test-key", - reasoning: "minimal", - signal: createAbortedSignal(), - onPayload: payload => resolve(payload), - }); - const payload = await promise; - const thinking = getNestedObject(payload, "thinking"); - - expect(thinking?.type).toBe("disabled"); - expect(toObject(payload)?.reasoning_effort).toBeUndefined(); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); }); it("treats finish_reason end as stop", async () => { @@ -1217,7 +1206,7 @@ describe("kimi model detection via detectCompat", () => { expect(openRouterKimi.compat.thinkingFormat).toBe("openrouter"); }); - it("maps OpenRouter Anthropic adaptive reasoning efforts to the Anthropic scale", async () => { + it("sends OpenRouter Anthropic adaptive reasoning efforts 1:1 on the wire", async () => { const model: Model<"openai-completions"> = buildModel({ ...gpt4oMiniSpec, api: "openai-completions", @@ -1229,9 +1218,11 @@ describe("kimi model detection via detectCompat", () => { const highPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "high" }); const xhighPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "xhigh" }); + const maxPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "max" }); - expect(getNestedObject(highPayload, "reasoning")).toEqual({ effort: "xhigh" }); - expect(getNestedObject(xhighPayload, "reasoning")).toEqual({ effort: "max" }); + expect(getNestedObject(highPayload, "reasoning")).toEqual({ effort: "high" }); + expect(getNestedObject(xhighPayload, "reasoning")).toEqual({ effort: "xhigh" }); + expect(getNestedObject(maxPayload, "reasoning")).toEqual({ effort: "max" }); }); // Regression for #1071: OpenCode-Go/Zen handle reasoning content server-side diff --git a/packages/ai/test/openai-reasoning-effort-fallback.test.ts b/packages/ai/test/openai-reasoning-effort-fallback.test.ts index 1676e29ff..991be7df8 100644 --- a/packages/ai/test/openai-reasoning-effort-fallback.test.ts +++ b/packages/ai/test/openai-reasoning-effort-fallback.test.ts @@ -171,10 +171,10 @@ function createResponsesModel(): Model<"openai-responses"> { maxTokens: 16_384, }); } -function createMappedResponsesModel(): Model<"openai-responses"> { +function createMaxLadderResponsesModel(): Model<"openai-responses"> { return buildModel({ - id: "mapped-responses-reasoner", - name: "Mapped Responses Reasoner", + id: "max-ladder-responses-reasoner", + name: "Max Ladder Responses Reasoner", api: "openai-responses", provider: "custom-responses", baseUrl: "https://responses.example.test/v1", @@ -185,8 +185,7 @@ function createMappedResponsesModel(): Model<"openai-responses"> { }, thinking: { mode: "effort", - efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }, input: ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, @@ -285,10 +284,10 @@ describe("OpenAI reasoning effort fallback retry", () => { { preconnect: fetch.preconnect }, ); - const result = await streamOpenAIResponses(createMappedResponsesModel(), testContext, { + const result = await streamOpenAIResponses(createMaxLadderResponsesModel(), testContext, { apiKey: "test-key", fetch: fetchMock, - reasoning: "xhigh", + reasoning: "max", }).result(); expect(result.stopReason).toBe("stop"); diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index ce1769430..c48f90d83 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -1765,7 +1765,7 @@ describe("Generate E2E Tests", () => { tools: [calculatorTool], }, { - reasoning: Effort.XHigh, + reasoning: Effort.Max, interleavedThinking: true, onPayload: payload => { capturedPayload = payload; diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 1df563b14..02e7eee1e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -10,10 +10,17 @@ - Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) - Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). - Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. +- Added `Effort.Max` ("max") as a first-class user-facing thinking level above `xhigh`. ### Changed +- Standardized reasoning effort levels to use a wire-exact `max` tier across all model providers +- Refactored Devin model routing to support 1:1 mapping for the `max` effort tier +- Normalized stale Ollama model configurations to the new wire-exact effort ladder + - Updated costs and context windows for various models in the catalog +- **Breaking**: Effort ladders are now wire-exact and the shifted five-tier effort mapping is gone. Models expose exactly the effort tiers their wire accepts, mapped 1:1: GPT-5.6+ and Anthropic adaptive models with a real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5) expose `low..max`; legacy adaptive models (Opus 4.6 and all Bedrock adaptive) expose `low/medium/high/max`; Sonnet/Haiku 4.6 expose `low/medium/high`; GLM-5.2 on Z.ai/Zhipu/Umans/Ollama Cloud/Baseten and Sakana Fugu and DeepSeek expose `high/max`; local Ollama reasoning models expose `low/medium/high/max`; Fire Pass Kimi exposes `low..max` with distinct `xhigh` and `max` budgets. Removed `SHIFTED_FIVE_TIER_EFFORT_MAP`, `ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER`, and the per-host `xhigh -> "max"` alias maps; selecting a tier a model lacks clamps down via `clampThinkingLevelForModel`. Devin effort routing is now 1:1 onto per-tier siblings (`max -> -max`; families without a `-max` sibling top out at `xhigh`; the fake `minimal -> -low` fallback is gone). Regenerated `models.json`. +- Changed `fillThinkingWireDefaults` to re-derive the effort map whenever the model-defined ladder disagrees with cached metadata, so stale cached surfaces from the shifted-map era normalize to the new wire-exact shape on every `buildModel`. ## [16.3.15] - 2026-07-09 diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index dcfcd3df9..a01430b33 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -71,31 +71,11 @@ const DSML_HEALING_PROVIDERS = new Set([ "openrouter", ]); -/** - * Ollama's OpenAI-compatible `reasoning.effort` only accepts - * `high|medium|low|max|none`; OMP's `minimal`/`xhigh` levels make the server - * reject the turn with HTTP 400 `invalid reasoning value`. Map the two - * unsupported levels onto the closest accepted ones. Stamped in the compat - * builder (not only at discovery) so stale-cached and custom `ollama`-provider - * specs are backfilled on every `buildModel`, not just on a fresh - * `omp models refresh`. Custom OpenAI-compatible providers pointed at a local - * Ollama port under a different provider id are not covered — they must set - * `compat.reasoningEffortMap` themselves. - */ -const OLLAMA_REASONING_EFFORT_MAP: ResolvedOpenAISharedCompat["reasoningEffortMap"] = { minimal: "low", xhigh: "max" }; - -/** - * Merge the Ollama default effort map under any explicit overrides (overrides - * win). No-op off the local `ollama` provider or for non-reasoning models. - */ -function mergeOllamaReasoningEffortMap( - compat: ResolvedOpenAISharedCompat, - provider: string, - reasoning: boolean | undefined, -): void { - if (provider !== "ollama" || !reasoning) return; - compat.reasoningEffortMap = { ...OLLAMA_REASONING_EFFORT_MAP, ...compat.reasoningEffortMap }; -} +// Ollama's OpenAI-compatible `reasoning.effort` accepts `high|medium|low|max|none`; +// `ollama`-provider reasoning models carry the wire-exact `low..max` effort +// ladder (see getModelDefinedEfforts), so no compat-level remapping is needed. +// Custom OpenAI-compatible providers pointed at a local Ollama port under a +// different provider id must set `compat.reasoningEffortMap` themselves. function resolveReasoningDisableMode( thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"], @@ -554,7 +534,6 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(compat, provider, spec.reasoning); mergeMimoReasoningEffortMap(compat, isMimoReasoningEffortModel); const whenThinkingPolicy = @@ -568,7 +547,6 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) { variant.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(variant, provider, spec.reasoning); mergeMimoReasoningEffortMap(variant, isMimoReasoningEffortModel); compat.whenThinking = variant; } @@ -678,7 +656,6 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; } - mergeOllamaReasoningEffortMap(compat, spec.provider, spec.reasoning); return compat; } diff --git a/packages/catalog/src/effort.ts b/packages/catalog/src/effort.ts index 831a13ede..e3491c2a9 100644 --- a/packages/catalog/src/effort.ts +++ b/packages/catalog/src/effort.ts @@ -5,6 +5,7 @@ export const enum Effort { Medium = "medium", High = "high", XHigh = "xhigh", + Max = "max", } export const THINKING_EFFORTS: readonly Effort[] = [ @@ -13,4 +14,5 @@ export const THINKING_EFFORTS: readonly Effort[] = [ Effort.Medium, Effort.High, Effort.XHigh, + Effort.Max, ]; diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 611eb1f60..f2842e4d5 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -61,12 +61,34 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High]; const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High]; -const GLM_52_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.XHigh]; - -const FUGU_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.XHigh]; -const FUGU_REASONING_EFFORT_MAP: Readonly = { - [Effort.XHigh]: "max", -}; +/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek. */ +const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max]; +/** OpenRouter's DeepSeek route accepts only `high`. */ +const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High]; +/** + * Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models + * with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the + * Fire Pass Kimi router (distinct xhigh and max budgets). + */ +const FIVE_TIER_EFFORTS_LOW_TO_MAX: readonly Effort[] = [ + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, +]; +/** Legacy adaptive scale (Opus/Sonnet 4.6, every Bedrock adaptive model): four wire tiers, no xhigh. */ +const FOUR_TIER_EFFORTS_LOW_TO_MAX: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.Max]; +/** GLM-5.2 resellers that pass the default lower tiers verbatim and expose the genuine `max` top tier. */ +const DEFAULT_REASONING_EFFORTS_WITH_MAX: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.Max, +]; +/** Local Ollama wire vocabulary (`low`/`medium`/`high`/`max`; `none` is thinking-off). */ +const OLLAMA_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.Max]; type EffortMap = Partial>; const GROQ_QWEN3_32B_REASONING_EFFORT_MAP: Readonly = { @@ -76,57 +98,14 @@ const GROQ_QWEN3_32B_REASONING_EFFORT_MAP: Readonly = { [Effort.High]: "default", [Effort.XHigh]: "default", }; -const DEEPSEEK_REASONING_EFFORT_MAP: Readonly = { - [Effort.Minimal]: "high", - [Effort.Low]: "high", - [Effort.Medium]: "high", - [Effort.High]: "high", - [Effort.XHigh]: "max", -}; const FIREWORKS_REASONING_EFFORT_MAP: Readonly = { [Effort.Minimal]: "none", }; -const ZAI_GLM_52_REASONING_EFFORT_MAP: Readonly = { - [Effort.Minimal]: "none", - [Effort.Low]: "high", - [Effort.Medium]: "high", - [Effort.High]: "high", - [Effort.XHigh]: "max", -}; -const GLM_52_XHIGH_MAX_EFFORT_MAP: Readonly = { - [Effort.XHigh]: "max", -}; const MIMO_REASONING_EFFORT_MAP: Readonly = { [Effort.Minimal]: "low", [Effort.XHigh]: "high", }; -/** - * Effort → wire-value map for a shifted five-tier scale (`low..max`): - * user-facing efforts shift up one notch so the top tier reaches the genuine - * "max" and "high" lands on the recommended "xhigh" coding/agentic default. - * Used by Anthropic adaptive models with a real xhigh tier (Opus 4.7+ and - * Fable/Mythos 5 on the Messages API) and by GPT-5.6+ wire-effort models, - * which expose the same genuine `max` tier above `xhigh`. - */ -export const SHIFTED_FIVE_TIER_EFFORT_MAP: Readonly>> = { - [Effort.Minimal]: "low", - [Effort.Low]: "medium", - [Effort.Medium]: "high", - [Effort.High]: "xhigh", - [Effort.XHigh]: "max", -}; - -/** - * Effort → wire-value map for the legacy 4-tier adaptive scale (Opus 4.6, - * Sonnet 4.6+, and every adaptive model on Bedrock Converse). `low..high` pass - * through verbatim; there is no real "xhigh", so it aliases the top "max" tier. - */ -export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER: Readonly>> = { - [Effort.Minimal]: "low", - [Effort.XHigh]: "max", -}; - const MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP: Readonly = { [Effort.Low]: "adaptive", [Effort.Medium]: "adaptive", @@ -172,8 +151,10 @@ export function resolveModelThinking( /** * Backfill identity-derived wire facts onto explicit thinking metadata. * Explicit `effortMap` / `supportsDisplay` (including `false`) win, except - * model-defined effort restrictions still normalize stale cached capability - * surfaces before request-time code can observe them. + * when the model-defined effort ladder disagrees with the cached surface: + * then both the ladder AND the wire map are re-derived from identity, so + * stale cached metadata from before a wire-truth change (e.g. the retired + * shifted five-tier maps) cannot survive normalization. */ function fillThinkingWireDefaults( spec: ModelSpec, @@ -184,11 +165,9 @@ function fillThinkingWireDefaults( const normalizedEfforts = getModelDefinedEfforts(spec, compat) ?? thinking.efforts; const effortsChanged = !sameEffortList(normalizedEfforts, thinking.efforts); const effortMap = - thinking.effortMap === undefined - ? inferEffortMap(spec, compat, parsed, thinking.mode, normalizedEfforts) - : effortsChanged - ? filterEffortMapToSupportedEfforts(thinking.effortMap, normalizedEfforts) - : undefined; + thinking.effortMap === undefined || effortsChanged + ? inferEffortMap(spec, compat, thinking.mode, normalizedEfforts) + : undefined; const shouldReplaceEffortMap = thinking.effortMap === undefined ? effortMap !== undefined : effortsChanged; const needsDisplay = thinking.supportsDisplay === undefined && @@ -229,7 +208,7 @@ export function deriveThinking(spec: ModelSpec, compat: mode: inferThinkingControlMode(spec, parsed), efforts, }; - const effortMap = inferEffortMap(spec, compat, parsed, config.mode, config.efforts); + const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts); if (effortMap !== undefined) { config.effortMap = effortMap; } @@ -264,11 +243,10 @@ function omitsWireReasoningEffort(api: Api, compat: CompatOf): boolean { function inferEffortMap( spec: ModelSpec, compat: CompatOf, - parsedModel: ParsedModel, mode: ThinkingConfig["mode"], efforts: readonly Effort[], ): EffortMap | undefined { - const detected = inferDetectedEffortMap(spec, compat, parsedModel, mode); + const detected = inferDetectedEffortMap(spec, compat, mode); const configured = readCompatEffortMap(compat); const merged = detected === undefined ? configured : configured === undefined ? detected : { ...detected, ...configured }; @@ -300,9 +278,9 @@ function isOpenAICompatReasoningApi(api: Api): boolean { /** * GPT-5.6+ addressed through a wire `reasoning.effort`/`reasoning_effort` - * field, where the shifted five-tier map applies. Devin (`devin-agent`) - * selects effort by routing to per-tier sibling model ids instead and must - * stay unmapped. + * field, where the five-tier `low..max` wire scale applies. Devin + * (`devin-agent`) selects effort by routing to per-tier sibling model ids + * instead and must stay unmapped. */ function isGpt56PlusWireEffortModel(spec: ModelSpec): boolean { switch (spec.api) { @@ -324,24 +302,64 @@ function getModelDefinedEfforts( compat: CompatOf, ): readonly Effort[] | undefined { if (isGlm52ReasoningEffortModelId(spec.id)) { - // Z.ai/Zhipu and OpenRouter both surface GLM-5.2's full effort ladder, - // including the top `xhigh` (= "max") tier; Umans and Ollama Cloud - // expose only high/max. - if (isZaiThinkingFormat(compat) || isOpenRouterThinkingFormat(compat)) { + // GLM-5.2's reasoning_effort dialect is host-specific (verified against + // live endpoints): + // - Z.ai/Zhipu ("zai" dialect) expose only high/max ("none" is the + // thinking-off state, not a user tier). + // - Umans, Ollama Cloud, and Baseten serve the same two-tier + // high/max scale on their GLM-5.2 routes. + // - OpenRouter rejects `max` — `xhigh` IS its top tier. + // - Other openai-compat hosts (Fireworks, resellers) pass the + // default lower tiers through verbatim and expose the genuine + // `max` above `high` (host quirks like Fireworks' minimal→none + // stay in the host maps). + if (isOpenRouterThinkingFormat(compat)) { return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; } - if (isUmansGlm52ReasoningEffortModel(spec) || isOllamaCloudGlm52ReasoningEffortModel(spec)) { - return GLM_52_HIGH_MAX_REASONING_EFFORTS; + if ( + isZaiThinkingFormat(compat) || + isUmansGlm52ReasoningEffortModel(spec) || + isOllamaCloudGlm52ReasoningEffortModel(spec) || + spec.provider === "baseten" + ) { + return HIGH_MAX_REASONING_EFFORTS; + } + if (isOpenAICompatReasoningApi(spec.api)) { + return DEFAULT_REASONING_EFFORTS_WITH_MAX; } } if (isSakanaFuguReasoningModel(spec)) { - return FUGU_REASONING_EFFORTS; + return HIGH_MAX_REASONING_EFFORTS; } if (isGpt56PlusWireEffortModel(spec)) { - // Normalize stale baked/discovered `low..xhigh` surfaces to the full - // five-tier ladder so the shifted map keeps the native `low` tier - // reachable (user `minimal`). - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + // Normalize stale baked/discovered `low..xhigh` surfaces to the + // wire-exact five-tier `low..max` ladder. + return FIVE_TIER_EFFORTS_LOW_TO_MAX; + } + const anthropicAdaptive = getAnthropicAdaptiveEfforts(spec); + if (anthropicAdaptive !== undefined) { + return anthropicAdaptive; + } + // Fire Pass's Kimi router accepts low..max with distinct xhigh and max + // budgets; user minimal has no wire tier there. + if (spec.provider === "firepass") { + return FIVE_TIER_EFFORTS_LOW_TO_MAX; + } + // Local Ollama's effort vocabulary is low/medium/high/max regardless of + // model. Custom OpenAI-compatible providers pointed at an Ollama port + // under a different provider id must set `compat.reasoningEffortMap` + // themselves. + if (spec.provider === "ollama") { + return OLLAMA_REASONING_EFFORTS; + } + if (isOpenAICompatReasoningApi(spec.api) && isDeepseekReasoningModel(spec)) { + // DeepSeek's reasoning_effort accepts only high/max; OpenRouter's + // DeepSeek route tops out at high. + return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS; + } + if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) { + // Baseten's gpt-oss router mirrors its GLM route: high/max only. + return HIGH_MAX_REASONING_EFFORTS; } return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || @@ -351,6 +369,27 @@ function getModelDefinedEfforts( : undefined; } +/** + * Wire-exact effort ladders for Anthropic adaptive models (4.6+). Model-defined + * so stale cached surfaces normalize on every build: Messages-API models with + * the real xhigh tier (4.7+) expose the full five-tier `low..max` scale; + * Opus/Sonnet 4.6 and every Bedrock adaptive model stay on the four-tier + * `low/medium/high/max` scale. + */ +function getAnthropicAdaptiveEfforts(spec: ModelSpec): readonly Effort[] | undefined { + const parsed = parseAnthropicModel(bareModelId(spec.id)); + if (!parsed || !isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; + if (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") { + return anthropicModelHasRealXHighEffort(spec, parsed) + ? FIVE_TIER_EFFORTS_LOW_TO_MAX + : FOUR_TIER_EFFORTS_LOW_TO_MAX; + } + if (isOpenRouterAnthropicAdaptiveReasoningModel(parsed, spec)) { + return isAnthropicAdaptiveGenAtLeast(parsed, "4.7") ? FIVE_TIER_EFFORTS_LOW_TO_MAX : FOUR_TIER_EFFORTS_LOW_TO_MAX; + } + return undefined; +} + function isOllamaCloudGlm52ReasoningEffortModel(spec: ModelSpec): boolean { return spec.api === "ollama-chat" && spec.provider === "ollama-cloud" && isGlm52ReasoningEffortModelId(spec.id); } @@ -395,62 +434,31 @@ function isZaiThinkingFormat(compat: CompatOf): boolean { function inferDetectedEffortMap( spec: ModelSpec, compat: CompatOf, - parsedModel: ParsedModel, mode: ThinkingConfig["mode"], ): EffortMap | undefined { if (mode === "anthropic-adaptive") { if (isMinimaxReasoningModelOnAnthropicEndpoint(spec)) { return MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP; } - return anthropicModelHasRealXHighEffort(spec, parsedModel) - ? SHIFTED_FIVE_TIER_EFFORT_MAP - : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; - } - // GLM-5.2 coding SKUs accept `reasoning_effort`, but the effort dialect is - // host-specific (verified against live endpoints): - // - Z.ai/Zhipu ("zai" dialect): the model exposes only none/high/max, so - // `xhigh` 400s — collapse minimal->none, low/medium/high->high, xhigh->max. - // - OpenRouter: `max` 400s and `xhigh` IS its max tier, so it passes `xhigh` - // through literally (no map; the tier is exposed via getModelDefinedEfforts). - // - Umans and Ollama Cloud expose only high/max on their GLM-5.2 routes. - // - Other openai-compat hosts (Fireworks, resellers) keep their distinct - // lower tiers and host quirks (e.g. Fireworks rejects `minimal`, so - // `minimal->none` stays) and only remap the top `xhigh` UI tier onto the - // genuine `max` budget. Filtered to supported efforts later. - const isGlm52 = isGlm52ReasoningEffortModelId(spec.id); - if (isGlm52 && isZaiThinkingFormat(compat)) { - return ZAI_GLM_52_REASONING_EFFORT_MAP; - } - if (isUmansGlm52ReasoningEffortModel(spec) || isOllamaCloudGlm52ReasoningEffortModel(spec)) { - return GLM_52_XHIGH_MAX_EFFORT_MAP; - } - if (isSakanaFuguReasoningModel(spec)) { - return FUGU_REASONING_EFFORT_MAP; - } - if (isGpt56PlusWireEffortModel(spec)) { - return SHIFTED_FIVE_TIER_EFFORT_MAP; + // Adaptive effort ladders are wire-exact (see + // getAnthropicAdaptiveEfforts) — no mapping needed. + return undefined; } if (!isOpenAICompatReasoningApi(spec.api)) { return undefined; } - let map: EffortMap | undefined; if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") { - map = GROQ_QWEN3_32B_REASONING_EFFORT_MAP; - } else if (isDeepseekReasoningModel(spec)) { - map = DEEPSEEK_REASONING_EFFORT_MAP; - } else if (isOpenAICompatMimoReasoningEffortModel(spec, compat)) { - map = MIMO_REASONING_EFFORT_MAP; - } else if (modelMatchesHost(spec, "openrouter")) { - map = getOpenRouterAnthropicReasoningEffortMap(spec.id); - } else if (modelMatchesHost(spec, "fireworks")) { - map = FIREWORKS_REASONING_EFFORT_MAP; + return GROQ_QWEN3_32B_REASONING_EFFORT_MAP; } - // Overlay GLM-5.2's top-tier `xhigh -> max` on the host base map, except on - // OpenRouter (xhigh IS its max tier; `max` 400s there). - if (isGlm52 && !isOpenRouterThinkingFormat(compat)) { - map = { ...map, ...GLM_52_XHIGH_MAX_EFFORT_MAP }; + if (isOpenAICompatMimoReasoningEffortModel(spec, compat)) { + return MIMO_REASONING_EFFORT_MAP; } - return map; + // Host quirk: Fireworks rejects `minimal` (maps to `none`) on ladders + // that genuinely include it. Filtered to supported efforts later. + if (modelMatchesHost(spec, "fireworks")) { + return FIREWORKS_REASONING_EFFORT_MAP; + } + return undefined; } function isSakanaFuguReasoningModel(spec: ModelSpec): boolean { @@ -471,17 +479,6 @@ function isDeepseekReasoningModel(spec: ModelSpec): bool ); } -function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | undefined { - const parsed = parseAnthropicModel(bareModelId(modelId)); - if (!parsed) return undefined; - // Adaptive efforts on OpenRouter's completions front: Fable/Mythos, Sonnet 5+, - // and Opus 4.6+ only — older Sonnet versions stay on the plain effort vocabulary there. - if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; - - const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); - return hasRealXHigh ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; -} - function inferSupportedEfforts( parsedModel: ParsedModel, spec: ModelSpec, @@ -507,10 +504,9 @@ function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] { if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) { return GPT_5_1_CODEX_MINI_EFFORTS; } - // 5.6+ exposes the full five-tier ladder: the shifted wire map spans - // low..max, with user `minimal` reaching the native `low` tier. + // 5.6+ exposes the wire-exact five-tier ladder low..max. if (semverGte(model.version, "5.6")) { - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + return FIVE_TIER_EFFORTS_LOW_TO_MAX; } if (semverGte(model.version, "5.2")) { return GPT_5_2_PLUS_EFFORTS; @@ -554,16 +550,18 @@ function inferAnthropicSupportedEfforts( spec: ModelSpec, compat: CompatOf, ): readonly Effort[] { - if ( - (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && - semverGte(parsedModel.version, "4.6") - ) { - return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6") - ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH - : DEFAULT_REASONING_EFFORTS; + // Ladders for adaptive-generation models (Opus 4.6+, Sonnet 5+, + // Fable/Mythos) are model-defined and already resolved by + // getAnthropicAdaptiveEfforts. Every other 4.6+ model on the Messages + // API (Sonnet/Haiku 4.6) still runs adaptive mode with the three-tier + // low/medium/high wire scale — no minimal, no max. + if (spec.api === "anthropic-messages" && semverGte(parsedModel.version, "4.6")) { + return LOW_MEDIUM_HIGH_REASONING_EFFORTS; } - if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, spec)) { - return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + // Non-adaptive 4.6 models on Bedrock stay budget-mode, where minimal is + // a legitimate synthetic budget tier. + if (spec.api === "bedrock-converse-stream" && semverGte(parsedModel.version, "4.6")) { + return DEFAULT_REASONING_EFFORTS; } return inferFallbackEfforts(spec, compat); } @@ -752,6 +750,7 @@ export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW return "MEDIUM"; case Effort.High: case Effort.XHigh: + case Effort.Max: return "HIGH"; } } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index b19a10fd5..474e62114 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -1573,8 +1573,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 123000, + "maxTokens": 16000 }, "baidu/ernie-5-0-thinking-latest": { "id": "baidu/ernie-5-0-thinking-latest", @@ -2409,19 +2409,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -2445,19 +2435,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-chat-v3-0324": { @@ -2500,19 +2480,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1": { @@ -2536,19 +2506,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "elevenlabs/eleven_multilingual_v2": { @@ -7738,16 +7698,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic.claude-opus-4-7": { @@ -7772,16 +7727,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7807,16 +7757,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7842,16 +7787,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -7906,16 +7846,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "au.anthropic.claude-opus-4-8": { @@ -7940,16 +7875,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8033,16 +7963,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8330,16 +8255,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8481,16 +8401,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "eu.anthropic.claude-opus-4-7": { @@ -8515,16 +8430,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8550,16 +8460,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8672,16 +8577,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8736,16 +8636,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8829,16 +8724,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "global.anthropic.claude-opus-4-7": { @@ -8863,16 +8753,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -8898,16 +8783,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9020,16 +8900,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9124,16 +8999,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9159,16 +9029,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -9252,16 +9117,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10256,16 +10116,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10407,16 +10262,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "us.anthropic.claude-opus-4-7": { @@ -10441,16 +10291,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10476,16 +10321,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -10598,16 +10438,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11141,19 +10976,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11239,19 +11067,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11447,16 +11268,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7": { @@ -11481,19 +11297,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11519,19 +11328,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -11667,14 +11469,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -11699,19 +11497,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } } @@ -12679,19 +12470,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "moonshotai/Kimi-K2.5": { @@ -12863,9 +12644,8 @@ "thinking": { "mode": "effort", "efforts": [ - "low", - "medium", - "high" + "high", + "max" ] } }, @@ -13293,19 +13073,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13451,16 +13224,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4-7": { @@ -13485,19 +13253,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13523,19 +13284,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -13621,14 +13375,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -13653,19 +13403,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -14296,19 +14039,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Pro": { @@ -14332,19 +14065,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemma-4-31B-it": { @@ -16059,19 +15782,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -16109,19 +15822,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } } }, @@ -16328,19 +16031,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-7-low", - "low": "claude-opus-4-7-medium", - "medium": "claude-opus-4-7-high", - "high": "claude-opus-4-7-xhigh", - "xhigh": "claude-opus-4-7-max" + "low": "claude-opus-4-7-low", + "medium": "claude-opus-4-7-medium", + "high": "claude-opus-4-7-high", + "xhigh": "claude-opus-4-7-xhigh", + "max": "claude-opus-4-7-max" } }, "requestModelId": "claude-opus-4-7-low" @@ -16368,19 +16071,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-7-low-fast", - "low": "claude-opus-4-7-medium-fast", - "medium": "claude-opus-4-7-high-fast", - "high": "claude-opus-4-7-xhigh-fast", - "xhigh": "claude-opus-4-7-max-fast" + "low": "claude-opus-4-7-low-fast", + "medium": "claude-opus-4-7-medium-fast", + "high": "claude-opus-4-7-high-fast", + "xhigh": "claude-opus-4-7-xhigh-fast", + "max": "claude-opus-4-7-max-fast" } }, "requestModelId": "claude-opus-4-7-low-fast" @@ -16408,19 +16111,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-8-low", - "low": "claude-opus-4-8-medium", - "medium": "claude-opus-4-8-high", - "high": "claude-opus-4-8-xhigh", - "xhigh": "claude-opus-4-8-max" + "low": "claude-opus-4-8-low", + "medium": "claude-opus-4-8-medium", + "high": "claude-opus-4-8-high", + "xhigh": "claude-opus-4-8-xhigh", + "max": "claude-opus-4-8-max" } }, "requestModelId": "claude-opus-4-8-low" @@ -16448,19 +16151,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "requiresEffort": true, "effortRouting": { - "minimal": "claude-opus-4-8-low-fast", - "low": "claude-opus-4-8-medium-fast", - "medium": "claude-opus-4-8-high-fast", - "high": "claude-opus-4-8-xhigh-fast", - "xhigh": "claude-opus-4-8-max-fast" + "low": "claude-opus-4-8-low-fast", + "medium": "claude-opus-4-8-medium-fast", + "high": "claude-opus-4-8-high-fast", + "xhigh": "claude-opus-4-8-xhigh-fast", + "max": "claude-opus-4-8-max-fast" } }, "requestModelId": "claude-opus-4-8-low-fast" @@ -16917,7 +16620,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -16925,7 +16627,6 @@ ], "effortRouting": { "off": "MODEL_GPT_5_2_NONE", - "minimal": "MODEL_GPT_5_2_LOW", "low": "MODEL_GPT_5_2_LOW", "medium": "MODEL_GPT_5_2_MEDIUM", "high": "MODEL_GPT_5_2_HIGH", @@ -16957,7 +16658,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -16965,7 +16665,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-3-codex-low", "low": "gpt-5-3-codex-low", "medium": "gpt-5-3-codex-medium", "high": "gpt-5-3-codex-high", @@ -16997,7 +16696,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17005,7 +16703,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-3-codex-low-priority", "low": "gpt-5-3-codex-low-priority", "medium": "gpt-5-3-codex-medium-priority", "high": "gpt-5-3-codex-high-priority", @@ -17037,7 +16734,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17045,7 +16741,6 @@ ], "effortRouting": { "off": "gpt-5-4-none", - "minimal": "gpt-5-4-low", "low": "gpt-5-4-low", "medium": "gpt-5-4-medium", "high": "gpt-5-4-high", @@ -17077,7 +16772,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17085,7 +16779,6 @@ ], "effortRouting": { "off": "gpt-5-4-none-priority", - "minimal": "gpt-5-4-low-priority", "low": "gpt-5-4-low-priority", "medium": "gpt-5-4-medium-priority", "high": "gpt-5-4-high-priority", @@ -17117,7 +16810,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17125,7 +16817,6 @@ ], "requiresEffort": true, "effortRouting": { - "minimal": "gpt-5-4-mini-low", "low": "gpt-5-4-mini-low", "medium": "gpt-5-4-mini-medium", "high": "gpt-5-4-mini-high", @@ -17157,7 +16848,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17165,7 +16855,6 @@ ], "effortRouting": { "off": "gpt-5-5-none", - "minimal": "gpt-5-5-low", "low": "gpt-5-5-low", "medium": "gpt-5-5-medium", "high": "gpt-5-5-high", @@ -17197,7 +16886,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17205,7 +16893,6 @@ ], "effortRouting": { "off": "gpt-5-5-none-priority", - "minimal": "gpt-5-5-low-priority", "low": "gpt-5-5-low-priority", "medium": "gpt-5-5-medium-priority", "high": "gpt-5-5-high-priority", @@ -17237,19 +16924,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-luna-none", - "minimal": "gpt-5-6-luna-low", - "low": "gpt-5-6-luna-medium", - "medium": "gpt-5-6-luna-high", - "high": "gpt-5-6-luna-xhigh", - "xhigh": "gpt-5-6-luna-max" + "low": "gpt-5-6-luna-low", + "medium": "gpt-5-6-luna-medium", + "high": "gpt-5-6-luna-high", + "xhigh": "gpt-5-6-luna-xhigh", + "max": "gpt-5-6-luna-max" } }, "requestModelId": "gpt-5-6-luna-none" @@ -17277,7 +16964,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17285,7 +16971,6 @@ ], "effortRouting": { "off": "gpt-5-6-luna-none-priority", - "minimal": "gpt-5-6-luna-low-priority", "low": "gpt-5-6-luna-low-priority", "medium": "gpt-5-6-luna-medium-priority", "high": "gpt-5-6-luna-high-priority", @@ -17317,19 +17002,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-sol-none", - "minimal": "gpt-5-6-sol-low", - "low": "gpt-5-6-sol-medium", - "medium": "gpt-5-6-sol-high", - "high": "gpt-5-6-sol-xhigh", - "xhigh": "gpt-5-6-sol-max" + "low": "gpt-5-6-sol-low", + "medium": "gpt-5-6-sol-medium", + "high": "gpt-5-6-sol-high", + "xhigh": "gpt-5-6-sol-xhigh", + "max": "gpt-5-6-sol-max" } }, "requestModelId": "gpt-5-6-sol-none" @@ -17357,7 +17042,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17365,7 +17049,6 @@ ], "effortRouting": { "off": "gpt-5-6-sol-none-priority", - "minimal": "gpt-5-6-sol-low-priority", "low": "gpt-5-6-sol-low-priority", "medium": "gpt-5-6-sol-medium-priority", "high": "gpt-5-6-sol-high-priority", @@ -17397,19 +17080,19 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "effortRouting": { "off": "gpt-5-6-terra-none", - "minimal": "gpt-5-6-terra-low", - "low": "gpt-5-6-terra-medium", - "medium": "gpt-5-6-terra-high", - "high": "gpt-5-6-terra-xhigh", - "xhigh": "gpt-5-6-terra-max" + "low": "gpt-5-6-terra-low", + "medium": "gpt-5-6-terra-medium", + "high": "gpt-5-6-terra-high", + "xhigh": "gpt-5-6-terra-xhigh", + "max": "gpt-5-6-terra-max" } }, "requestModelId": "gpt-5-6-terra-none" @@ -17437,7 +17120,6 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", @@ -17445,7 +17127,6 @@ ], "effortRouting": { "off": "gpt-5-6-terra-none-priority", - "minimal": "gpt-5-6-terra-low-priority", "low": "gpt-5-6-terra-low-priority", "medium": "gpt-5-6-terra-medium-priority", "high": "gpt-5-6-terra-high-priority", @@ -17812,15 +17493,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "none" - } + "xhigh", + "max" + ] } } }, @@ -17846,19 +17524,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] }, "compat": { "supportsToolChoice": false, @@ -17886,19 +17554,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] }, "compat": { "supportsToolChoice": false, @@ -18059,11 +17717,10 @@ "low", "medium", "high", - "xhigh" + "max" ], "effortMap": { - "minimal": "none", - "xhigh": "max" + "minimal": "none" } } }, @@ -18563,19 +18220,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18675,16 +18325,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4.7": { @@ -18713,19 +18358,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18755,19 +18393,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -18865,14 +18496,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -18901,19 +18528,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -21962,16 +21582,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7@default": { @@ -21996,19 +21611,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22034,19 +21642,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22102,14 +21703,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5@default": { @@ -22134,19 +21731,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -22171,19 +21761,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.2-maas": { @@ -22207,19 +21787,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemini-2.5-flash": { @@ -22746,19 +22316,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemma2-9b-it": { @@ -23177,19 +22737,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-R1-0528": { @@ -23213,19 +22763,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -23268,19 +22808,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Flash": { @@ -23304,19 +22834,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V4-Pro": { @@ -23340,19 +22860,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemma-4-26B-A4B-it": { @@ -26130,8 +25640,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 123000, + "maxTokens": 16000 }, "baidu/qianfan-ocr-fast": { "id": "baidu/qianfan-ocr-fast", @@ -26540,19 +26050,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1": { @@ -26576,19 +26076,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -26612,19 +26102,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-distill-llama-70b": { @@ -26686,19 +26166,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.1-terminus:exacto": { @@ -26741,19 +26211,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -26777,19 +26237,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-speciale": { @@ -26832,19 +26282,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash:discounted": { @@ -26906,19 +26346,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro:discounted": { @@ -28554,8 +27984,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", @@ -28746,8 +28176,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 65535, + "maxTokens": 8000 }, "minimax/minimax-01": { "id": "minimax/minimax-01", @@ -34446,7 +33876,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", @@ -34513,7 +33943,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", @@ -35579,11 +35009,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5v-turbo": { @@ -40546,19 +39973,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -40639,19 +40056,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - }, "effortRouting": { "off": "deepseek-ai/deepseek-v3.2-exp", "minimal": "deepseek-ai/deepseek-v3.2-exp-thinking", @@ -40840,19 +40247,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-prover-v2-671b": { @@ -40933,19 +40330,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash": { @@ -40969,19 +40356,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash:thinking": { @@ -41005,19 +40382,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -41041,19 +40408,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-cheaper": { @@ -41077,19 +40434,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-cheaper:thinking": { @@ -41113,19 +40460,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro:thinking": { @@ -41149,19 +40486,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "dmind/dmind-1": { @@ -45719,8 +45046,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", @@ -45835,8 +45162,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 65535, + "maxTokens": 8000 }, "MiniMax-M1": { "id": "MiniMax-M1", @@ -46134,8 +45461,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 1000000, + "maxTokens": 40000 }, "miromind-ai/mirothinker-v1.5-235b": { "id": "miromind-ai/mirothinker-v1.5-235b", @@ -48640,19 +47967,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-luna-pro": { @@ -48696,19 +48016,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol-pro": { @@ -48752,19 +48065,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra-pro": { @@ -51265,8 +50571,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 8192, + "maxTokens": 32000 }, "Sao10K/L3.1-70B-Euryale-v2.2": { "id": "Sao10K/L3.1-70B-Euryale-v2.2", @@ -51994,19 +51300,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "TEE/deepseek-v4-pro:thinking": { @@ -52030,19 +51326,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "TEE/gemma-3-27b-it": { @@ -52317,11 +51603,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "TEE/gpt-oss-120b": { @@ -52768,7 +52051,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null + "maxTokens": 262144 }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", @@ -53024,8 +52307,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32000, + "maxTokens": 32000 }, "THUDM/GLM-4-9B-0414": { "id": "THUDM/GLM-4-9B-0414", @@ -54774,11 +54057,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "zai-org/glm-latest": { @@ -55017,19 +54297,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -55054,19 +54324,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528-qwen3-8b": { @@ -55111,19 +54371,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-turbo": { @@ -55148,19 +54398,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1/community": { @@ -55185,19 +54425,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3-0324": { @@ -55262,19 +54492,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.1-terminus": { @@ -55299,19 +54519,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2": { @@ -55336,19 +54546,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -55373,19 +54573,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3/community": { @@ -55430,19 +54620,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -55467,19 +54647,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "dev/glm46": { @@ -57615,11 +56785,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "zai-org/glm-5v-turbo": { @@ -57885,19 +57052,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.1": { @@ -57921,19 +57078,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.1-terminus": { @@ -57957,19 +57104,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v3.2": { @@ -57993,19 +57130,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v4-flash": { @@ -58029,19 +57156,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/deepseek-v4-pro": { @@ -58065,19 +57182,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/codegemma-1.1-7b": { @@ -58568,8 +57675,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 32000 }, "meta/llama-3.2-90b-vision-instruct": { "id": "meta/llama-3.2-90b-vision-instruct", @@ -61082,11 +60189,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm4.7": { @@ -61619,11 +60723,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "gpt-oss:120b": { @@ -63271,19 +62372,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna": { @@ -63309,19 +62403,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna-pro": { @@ -63343,26 +62430,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -63387,19 +62467,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-sol-pro": { @@ -63421,26 +62494,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -63465,19 +62531,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-terra-pro": { @@ -63499,26 +62558,19 @@ }, "contextWindow": 1050000, "maxTokens": 128000, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "o1": { "id": "o1", @@ -64330,19 +63382,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-luna-pro": { @@ -64372,26 +63417,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 3, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-luna", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", @@ -64424,19 +63462,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-sol-pro": { @@ -64466,26 +63497,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 1, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-sol", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", @@ -64518,19 +63542,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "gpt-5.6-terra-pro": { @@ -64560,26 +63577,19 @@ "preferWebsockets": true, "useResponsesLite": true, "priority": 2, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } - }, - "requestModelId": "gpt-5.6-terra", - "reasoningMode": "pro" + "xhigh", + "max" + ] + } } }, "opencode": { @@ -64757,19 +63767,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -64799,19 +63799,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "glm-5": { @@ -64897,11 +63887,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "kimi-k2.5": { @@ -65350,19 +64337,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "claude-3-5-haiku": { @@ -65407,19 +64384,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65535,16 +64505,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "claude-opus-4-7": { @@ -65569,19 +64534,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65607,19 +64565,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65705,14 +64656,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "claude-sonnet-5": { @@ -65737,19 +64684,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -65774,19 +64714,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-flash-free": { @@ -65810,19 +64740,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -65846,19 +64766,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "gemini-3-flash": { @@ -66118,11 +65028,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "gpt-5": { @@ -68094,19 +67001,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-haiku-4.5": { @@ -68247,16 +67147,11 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.6-fast": { @@ -68281,16 +67176,11 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -68315,19 +67205,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.7-fast": { @@ -68352,19 +67235,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.8": { @@ -68389,19 +67265,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-opus-4.8-fast": { @@ -68426,19 +67295,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "anthropic/claude-sonnet-4": { @@ -68550,19 +67412,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "arcee-ai/trinity-large-preview": { @@ -69091,17 +67946,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-r1": { @@ -69125,17 +67971,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-r1-0528": { @@ -69159,17 +67996,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.1-terminus": { @@ -69193,17 +68021,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.1-terminus:exacto": { @@ -69227,17 +68046,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.2": { @@ -69261,17 +68071,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -69295,17 +68096,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-flash": { @@ -69329,17 +68121,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-flash:free": { @@ -69363,17 +68146,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "deepseek/deepseek-v4-pro": { @@ -69397,17 +68171,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "essentialai/rnj-1-instruct": { @@ -73011,19 +71776,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-luna-pro": { @@ -73048,19 +71806,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol": { @@ -73085,19 +71836,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol-pro": { @@ -73122,19 +71866,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra": { @@ -73159,19 +71896,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra-pro": { @@ -73196,19 +71926,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-audio": { @@ -75714,7 +74437,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -75868,17 +74591,8 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } + ] } }, "tngtech/tng-r1t-chimera": { @@ -76815,13 +75529,13 @@ "text" ], "cost": { - "input": 0.84, - "output": 2.64, - "cacheRead": 0.156, + "input": 0.77, + "output": 2.42, + "cacheRead": 0.143, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 128000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -76876,19 +75590,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } } }, @@ -76956,11 +75660,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -76989,11 +75690,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -77022,11 +75720,8 @@ "mode": "effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "compat": { "includeEncryptedReasoning": false, @@ -77337,19 +76032,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3": { @@ -77392,19 +76077,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-ai/DeepSeek-V3.1": { @@ -77447,19 +76122,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "essentialai/Rnj-1-Instruct": { @@ -78224,11 +76889,8 @@ "mode": "anthropic-budget-effort", "efforts": [ "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] }, "input": [ "text" @@ -78863,19 +77525,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-flash": { @@ -78899,19 +77551,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -78935,19 +77577,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "e2ee-deepseek-v4-flash": { @@ -82488,19 +81120,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82646,16 +81271,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -82680,19 +81300,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82718,19 +81331,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -82816,14 +81422,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -82848,19 +81450,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -85843,11 +84438,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -85873,11 +84468,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -85903,11 +84498,11 @@ "thinking": { "mode": "budget", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } }, @@ -87470,19 +86065,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek-v4-pro": { @@ -87506,19 +86091,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "GLM-5.1": { @@ -88636,14 +87211,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -88656,6 +87223,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-4.3": { @@ -88677,14 +87252,6 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -88697,6 +87264,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-4.5": { @@ -88737,20 +87312,7 @@ }, "includeEncryptedReasoning": false, "filterReasoningHistory": true, - "omitReasoningEffort": false - }, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "low" - } + "omitReasoningEffort": true } }, "grok-build": { @@ -89905,19 +88467,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -89943,19 +88498,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90091,16 +88639,11 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "xhigh": "max" - } + "max" + ] } }, "anthropic/claude-opus-4.7": { @@ -90125,19 +88668,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90163,19 +88699,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90261,14 +88790,10 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high" - ], - "effortMap": { - "minimal": "low" - } + ] } }, "anthropic/claude-sonnet-5": { @@ -90293,19 +88818,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90331,19 +88849,12 @@ "thinking": { "mode": "anthropic-adaptive", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - }, "supportsDisplay": true } }, @@ -90776,19 +89287,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-r1-0528": { @@ -90812,19 +89313,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-reasoner": { @@ -90848,19 +89339,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" + "max" ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - }, "requiresEffort": true } }, @@ -90885,19 +89366,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v3.2-exp": { @@ -90921,19 +89392,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash": { @@ -90957,19 +89418,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-flash-free": { @@ -90993,19 +89444,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro": { @@ -91029,19 +89470,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "deepseek/deepseek-v4-pro-free": { @@ -91065,19 +89496,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "google/gemini-2.0-flash": { @@ -93131,19 +91552,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-sol": { @@ -93168,19 +91582,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-5.6-terra": { @@ -93205,19 +91612,12 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "low", - "low": "medium", - "medium": "high", - "high": "xhigh", - "xhigh": "max" - } + "xhigh", + "max" + ] } }, "openai/gpt-image-1.5": { @@ -94060,7 +92460,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": null, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -95169,11 +93569,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5.2-free": { @@ -95201,11 +93598,8 @@ "low", "medium", "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } + "max" + ] } }, "z-ai/glm-5v-turbo": { @@ -95516,19 +93910,9 @@ "thinking": { "mode": "effort", "efforts": [ - "minimal", - "low", - "medium", "high", - "xhigh" - ], - "effortMap": { - "minimal": "none", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } + "max" + ] } }, "glm-5v-turbo": { @@ -95566,4 +93950,4 @@ } } } -} +} \ No newline at end of file diff --git a/packages/catalog/src/provider-models/ollama.ts b/packages/catalog/src/provider-models/ollama.ts index 3708a8f0f..b9248903e 100644 --- a/packages/catalog/src/provider-models/ollama.ts +++ b/packages/catalog/src/provider-models/ollama.ts @@ -25,8 +25,7 @@ type OllamaShowResponse = { const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000]; const OLLAMA_CLOUD_GLM_52_THINKING: ThinkingConfig = { mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }; function trimTrailingSlash(value: string): string { diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 0b785b6ca..421a6a404 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -622,9 +622,8 @@ const UMANS_REASONING_EFFORT_BY_LEVEL: Record = { medium: Effort.Medium, high: Effort.High, xhigh: Effort.XHigh, - max: Effort.XHigh, + max: Effort.Max, }; -const UMANS_MAX_REASONING_EFFORT_MAP = { [Effort.XHigh]: "max" } as const; const UMANS_DEFAULT_REASONING_EFFORTS = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const; const UMANS_VIA_HANDOFF_MODEL_IDS = ["umans-glm-5.1", "umans-glm-5.2"] as const; @@ -689,9 +688,6 @@ function mapUmansThinkingConfig(value: unknown): ThinkingConfig | undefined { mode: umansHasMaxReasoningLevel(value) ? "anthropic-budget-effort" : "budget", efforts, }; - if (thinking.mode === "anthropic-budget-effort") { - thinking.effortMap = UMANS_MAX_REASONING_EFFORT_MAP; - } if (isRecord(value)) { if (value.can_disable === false) { thinking.requiresEffort = true; @@ -2802,18 +2798,13 @@ export function basetenModelManagerOptions( const baseModel = mapWithBundledReference(entry, defaults, reference); + // Baseten's reasoning router accepts only the high/max + // effort tiers for its GLM-5.2 and gpt-oss routes. const isEffortReasoning = defaults.id === "openai/gpt-oss-120b" || defaults.id === "zai-org/GLM-5.2"; const thinking = isEffortReasoning ? { mode: "effort" as const, - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], } : undefined; @@ -2936,8 +2927,7 @@ const SAKANA_FUGU_ULTRA_COST = { input: 5, output: 30, cacheRead: 0.5, cacheWrit const SAKANA_FUGU_ULTRA_CONTEXT_WINDOW = 1_000_000; const SAKANA_FUGU_THINKING: ThinkingConfig = { mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { [Effort.XHigh]: "max" }, + efforts: [Effort.High, Effort.Max], }; const SAKANA_RESPONSES_COMPAT: ModelSpec<"openai-responses">["compat"] = { includeEncryptedReasoning: false, diff --git a/packages/catalog/src/variant-collapse.ts b/packages/catalog/src/variant-collapse.ts index 46d1f9d52..770d7610d 100644 --- a/packages/catalog/src/variant-collapse.ts +++ b/packages/catalog/src/variant-collapse.ts @@ -111,15 +111,12 @@ function thinkingPair(baseId: string, name: string): EffortVariantFamily { }; } -type DevinTierRoutes = Partial>; +type DevinTierRoutes = Partial>; -const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, -]; +/** Devin families with a `-max` sibling: five wire tiers, `low` floor. */ +const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]; +/** Devin families topping out at `-xhigh` (pre-5.6 GPT, 5.6 fast lanes). */ +const DEVIN_FOUR_TIER_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; function devinTierFamily( id: string, @@ -132,11 +129,7 @@ function devinTierFamily( for (const effort of efforts) { switch (effort) { case Effort.Minimal: - if (routes.minimal) { - routing[effort] = routes.minimal; - } else if (routes.low) { - routing[effort] = routes.low; - } + if (routes.minimal) routing[effort] = routes.minimal; break; case Effort.Low: if (routes.low) routing[effort] = routes.low; @@ -150,11 +143,20 @@ function devinTierFamily( case Effort.XHigh: if (routes.xhigh) routing[effort] = routes.xhigh; break; + case Effort.Max: + if (routes.max) routing[effort] = routes.max; + break; } } - const members = [routes.off, routes.minimal, routes.low, routes.medium, routes.high, routes.xhigh].filter( - (member, index, items): member is string => typeof member === "string" && items.indexOf(member) === index, - ); + const members = [ + routes.off, + routes.minimal, + routes.low, + routes.medium, + routes.high, + routes.xhigh, + routes.max, + ].filter((member, index, items): member is string => typeof member === "string" && items.indexOf(member) === index); return { id, name, @@ -169,11 +171,9 @@ function devinTierFamily( } /** - * GPT-5.6 (Luna/Sol/Terra) adds a genuine `max` tier above `xhigh`, so the - * standard family shifts every user effort up one notch (`minimal` → `-low` - * … `xhigh` → `-max`), mirroring the Opus 4.7+ five-tier mapping. Devin - * serves no `-max-priority` sibling, so the fast family keeps the direct - * `low..xhigh` `-priority` scale. + * GPT-5.6 (Luna/Sol/Terra) serves per-tier siblings for the full five-tier + * `low..max` wire scale; user efforts route 1:1 onto them. Devin serves no + * `-max-priority` sibling, so the fast family tops out at `xhigh`. */ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): readonly EffortVariantFamily[] { const base = `gpt-5-6-${variant}`; @@ -183,11 +183,11 @@ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): re name, { off: `${base}-none`, - minimal: `${base}-low`, - low: `${base}-medium`, - medium: `${base}-high`, - high: `${base}-xhigh`, - xhigh: `${base}-max`, + low: `${base}-low`, + medium: `${base}-medium`, + high: `${base}-high`, + xhigh: `${base}-xhigh`, + max: `${base}-max`, }, DEVIN_FIVE_TIER_EFFORTS, ), @@ -201,7 +201,7 @@ function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): re high: `${base}-high-priority`, xhigh: `${base}-xhigh-priority`, }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), ]; } @@ -357,98 +357,54 @@ export const GEMINI_CLI_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }; export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { families: [ - { - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - members: [ - "claude-opus-4-7-low", - "claude-opus-4-7-medium", - "claude-opus-4-7-high", - "claude-opus-4-7-xhigh", - "claude-opus-4-7-max", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-7-low", - [Effort.Low]: "claude-opus-4-7-medium", - [Effort.Medium]: "claude-opus-4-7-high", - [Effort.High]: "claude-opus-4-7-xhigh", - [Effort.XHigh]: "claude-opus-4-7-max", + devinTierFamily( + "claude-opus-4-7", + "Claude Opus 4.7", + { + low: "claude-opus-4-7-low", + medium: "claude-opus-4-7-medium", + high: "claude-opus-4-7-high", + xhigh: "claude-opus-4-7-xhigh", + max: "claude-opus-4-7-max", }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-7-fast", + "Claude Opus 4.7 Fast", + { + low: "claude-opus-4-7-low-fast", + medium: "claude-opus-4-7-medium-fast", + high: "claude-opus-4-7-high-fast", + xhigh: "claude-opus-4-7-xhigh-fast", + max: "claude-opus-4-7-max-fast", }, - }, - { - id: "claude-opus-4-7-fast", - name: "Claude Opus 4.7 Fast", - members: [ - "claude-opus-4-7-low-fast", - "claude-opus-4-7-medium-fast", - "claude-opus-4-7-high-fast", - "claude-opus-4-7-xhigh-fast", - "claude-opus-4-7-max-fast", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-7-low-fast", - [Effort.Low]: "claude-opus-4-7-medium-fast", - [Effort.Medium]: "claude-opus-4-7-high-fast", - [Effort.High]: "claude-opus-4-7-xhigh-fast", - [Effort.XHigh]: "claude-opus-4-7-max-fast", + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-8", + "Claude Opus 4.8", + { + low: "claude-opus-4-8-low", + medium: "claude-opus-4-8-medium", + high: "claude-opus-4-8-high", + xhigh: "claude-opus-4-8-xhigh", + max: "claude-opus-4-8-max", }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + "claude-opus-4-8-fast", + "Claude Opus 4.8 Fast", + { + low: "claude-opus-4-8-low-fast", + medium: "claude-opus-4-8-medium-fast", + high: "claude-opus-4-8-high-fast", + xhigh: "claude-opus-4-8-xhigh-fast", + max: "claude-opus-4-8-max-fast", }, - }, - { - id: "claude-opus-4-8", - name: "Claude Opus 4.8", - members: [ - "claude-opus-4-8-low", - "claude-opus-4-8-medium", - "claude-opus-4-8-high", - "claude-opus-4-8-xhigh", - "claude-opus-4-8-max", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-8-low", - [Effort.Low]: "claude-opus-4-8-medium", - [Effort.Medium]: "claude-opus-4-8-high", - [Effort.High]: "claude-opus-4-8-xhigh", - [Effort.XHigh]: "claude-opus-4-8-max", - }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, - }, - }, - { - id: "claude-opus-4-8-fast", - name: "Claude Opus 4.8 Fast", - members: [ - "claude-opus-4-8-low-fast", - "claude-opus-4-8-medium-fast", - "claude-opus-4-8-high-fast", - "claude-opus-4-8-xhigh-fast", - "claude-opus-4-8-max-fast", - ], - routing: { - [Effort.Minimal]: "claude-opus-4-8-low-fast", - [Effort.Low]: "claude-opus-4-8-medium-fast", - [Effort.Medium]: "claude-opus-4-8-high-fast", - [Effort.High]: "claude-opus-4-8-xhigh-fast", - [Effort.XHigh]: "claude-opus-4-8-max-fast", - }, - thinking: { - mode: "effort", - efforts: DEVIN_FIVE_TIER_EFFORTS, - requiresEffort: true, - }, - }, + DEVIN_FIVE_TIER_EFFORTS, + ), devinTierFamily( "gpt-5-2", "GPT-5.2", @@ -459,7 +415,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "MODEL_GPT_5_2_HIGH", xhigh: "MODEL_GPT_5_2_XHIGH", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex", @@ -470,7 +426,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high", xhigh: "gpt-5-3-codex-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex-fast", @@ -481,7 +437,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high-priority", xhigh: "gpt-5-3-codex-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4", @@ -493,7 +449,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high", xhigh: "gpt-5-4-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-fast", @@ -505,7 +461,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high-priority", xhigh: "gpt-5-4-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-mini", @@ -516,7 +472,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-mini-high", xhigh: "gpt-5-4-mini-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5", @@ -528,7 +484,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high", xhigh: "gpt-5-5-xhigh", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5-fast", @@ -540,7 +496,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high-priority", xhigh: "gpt-5-5-xhigh-priority", }, - DEVIN_FIVE_TIER_EFFORTS, + DEVIN_FOUR_TIER_EFFORTS, ), ...devinGpt56Families("luna", "GPT-5.6 Luna"), ...devinGpt56Families("sol", "GPT-5.6 Sol"), diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index bd32f2f43..2c6da8a51 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -76,8 +76,7 @@ describe("generated model policies", () => { expect(models[0]?.cost.cacheWrite).toBe(6.25); expect(models[1]?.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { minimal: "low", xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], }); expect(models[1]?.cost.cacheRead).toBe(0.5); expect(models[1]?.cost.cacheWrite).toBe(6.25); @@ -103,8 +102,7 @@ describe("generated model policies", () => { expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 }); expect(models[0]?.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], supportsDisplay: true, }); }); diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index c353ac16f..48ded780f 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -277,10 +277,9 @@ describe("LiteLLM provider discovery", () => { reasoning: true, thinking: { mode: "effort", - efforts: ["minimal", "low", "medium", "high", "xhigh"], + efforts: ["minimal", "low", "medium", "high", "max"], effortMap: { minimal: "none", - xhigh: "max", }, }, }); @@ -709,10 +708,9 @@ describe("LiteLLM provider discovery", () => { reasoning: true, thinking: { mode: "effort", - efforts: ["minimal", "low", "medium", "high", "xhigh"], + efforts: ["minimal", "low", "medium", "high", "max"], effortMap: { minimal: "none", - xhigh: "max", }, }, }); diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index a21a323fc..ed08575f1 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -196,7 +196,7 @@ describe("model thinking derivation", () => { api: "openai-completions", provider: "deepseek", baseUrl: "https://api.deepseek.com/v1", - compat: { reasoningEffortMap: { xhigh: "max-plus" } }, + compat: { reasoningEffortMap: { max: "max-plus" } }, }); const openRouterAnthropic = createModel({ id: "anthropic/claude-opus-4.7", @@ -212,20 +212,20 @@ describe("model thinking derivation", () => { medium: "default", high: "default", }); - expect(deepseek.thinking?.effortMap).toMatchObject({ - minimal: "high", - low: "high", - medium: "high", - high: "high", - xhigh: "max-plus", - }); - expect(openRouterAnthropic.thinking?.effortMap).toEqual({ - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }); + // DeepSeek's ladder is the wire-exact high/max pair; explicit compat + // overrides still win over the identity wire values. + expect(getSupportedEfforts(deepseek)).toEqual([Effort.High, Effort.Max]); + expect(deepseek.thinking?.effortMap).toEqual({ max: "max-plus" }); + // OpenRouter-hosted Anthropic adaptive models carry the wire-exact + // five-tier ladder with no remapping. + expect(getSupportedEfforts(openRouterAnthropic)).toEqual([ + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, + ]); + expect(openRouterAnthropic.thinking?.effortMap).toBeUndefined(); }); it("maps GLM-5.2 reasoning effort per host dialect", () => { @@ -248,19 +248,20 @@ describe("model thinking derivation", () => { baseUrl: "https://openrouter.ai/api/v1", }); - // Z.ai dialect: the model only does none/high/max, so the lower tiers - // collapse and the top `xhigh` tier reaches `max`. - expect(zai.thinking?.effortMap).toEqual({ - minimal: "none", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }); - // Fireworks keeps its distinct lower tiers and the `minimal -> none` quirk; - // only the top `xhigh` UI tier remaps onto the genuine `max` budget. - expect(getSupportedEfforts(fireworks)).toContain(Effort.XHigh); - expect(fireworks.thinking?.effortMap).toEqual({ minimal: "none", xhigh: "max" }); + // Z.ai dialect: the model only does none/high/max on the wire, so the + // ladder is the honest high/max pair (none = thinking off). + expect(getSupportedEfforts(zai)).toEqual([Effort.High, Effort.Max]); + expect(zai.thinking?.effortMap).toBeUndefined(); + // Fireworks keeps its distinct lower tiers and the `minimal -> none` + // quirk; the genuine `max` tier sits above `high`. + expect(getSupportedEfforts(fireworks)).toEqual([ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.Max, + ]); + expect(fireworks.thinking?.effortMap).toEqual({ minimal: "none" }); // OpenRouter rejects `max` and treats `xhigh` as its max tier: expose the // `xhigh` tier and pass it through unmapped. expect(getSupportedEfforts(openRouter)).toContain(Effort.XHigh); @@ -428,30 +429,34 @@ describe("model thinking derivation", () => { }, }); expect(mapEffortToAnthropicAdaptiveEffort(minimaxM3, Effort.High)).toBe("adaptive"); - // Opus 4.6 has no real xhigh level — the baked 4-tier map aliases XHigh to "max". - expect(opus46.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" }); - expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toBe("max"); - // Opus 4.7+ on the Messages API exposes the full five-tier scale: the baked - // map shifts each user-facing effort up one notch so the top tier reaches "max". - expect(opus47.thinking?.effortMap).toEqual({ - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toBe("low"); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max"); - expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.High)).toBe("xhigh"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toBe("max"); - // Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max". - expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" }); + // Opus 4.6 has no real xhigh tier — the honest ladder is the four-tier + // low/medium/high/max wire scale, mapped 1:1. + expect(getSupportedEfforts(opus46)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(opus46.thinking?.effortMap).toBeUndefined(); + expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toThrow(/not supported/); + // Opus 4.7+ on the Messages API exposes the full five-tier wire scale + // low..max with no remapping. + expect(getSupportedEfforts(opus47)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus47.thinking?.effortMap).toBeUndefined(); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Low)).toBe("low"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("high"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toThrow(/not supported/); + expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.XHigh)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.Max)).toBe("max"); + // Bedrock Converse stays on the four-tier scale regardless of version. + expect(getSupportedEfforts(opus47Bedrock)).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(opus47Bedrock.thinking?.effortMap).toBeUndefined(); expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high"); - expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.High)).toBe("high"); + expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.Max)).toBe("max"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.Max)).toBe("max"); + expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toThrow(/not supported/); + // Sonnet 4.6 runs adaptive mode on the three-tier low/medium/high scale. + expect(getSupportedEfforts(sonnet46)).toEqual([Effort.Low, Effort.Medium, Effort.High]); expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/); + expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.Max)).toThrow(/not supported/); }); it("bakes adaptive display support for Opus 4.7+, Sonnet 5+, and Fable/Mythos 5", () => { @@ -488,8 +493,9 @@ describe("model thinking derivation", () => { }); it("backfills wire facts onto explicit thinking, explicit values winning", () => { - // Authored capability surface (mode/efforts) keeps identity-derived wire - // facts: configs never need to know Anthropic's tier tables. + // Authored partial ladders on wire-exact models normalize to the + // model-defined ladder, and the wire map is re-derived alongside: + // stale cached surfaces cannot pin retired wire facts. const filled = createModel({ id: "claude-opus-4-8", api: "anthropic-messages", @@ -498,24 +504,24 @@ describe("model thinking derivation", () => { }); expect(filled.thinking).toEqual({ mode: "anthropic-adaptive", - efforts: [Effort.Low, Effort.High], - effortMap: { low: "medium", high: "xhigh" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], supportsDisplay: true, }); - // Explicit wire facts are authoritative — including `false`. + // Explicit wire facts are authoritative — including `false` — when the + // authored ladder matches the wire truth. const pinned = createModel({ id: "claude-opus-4-8", api: "anthropic-messages", provider: "anthropic", thinking: { mode: "anthropic-adaptive", - efforts: [Effort.Low, Effort.High], - effortMap: { xhigh: "max" }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + effortMap: { max: "ultra" }, supportsDisplay: false, }, }); - expect(pinned.thinking?.effortMap).toEqual({ xhigh: "max" }); + expect(pinned.thinking?.effortMap).toEqual({ max: "ultra" }); expect(pinned.thinking?.supportsDisplay).toBe(false); }); @@ -568,7 +574,7 @@ describe("model thinking derivation", () => { expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined(); }); - it("bakes the GPT-5.6 shifted five-tier effort map on wire-effort APIs", () => { + it("bakes the wire-exact five-tier low..max ladder on GPT-5.6 wire-effort APIs", () => { const codex = createModel({ id: "gpt-5.6-sol", api: "openai-codex-responses", @@ -577,19 +583,12 @@ describe("model thinking derivation", () => { expect(codex.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }); - // Stale baked four-tier metadata (caches/discovery) normalizes back to - // the five-tier ladder with the map attached — the wire-defaults - // backfill path — and namespaced OpenRouter ids parse. + // Stale baked metadata (caches/discovery) — including shifted-era maps — + // normalizes to the wire-exact ladder with the map re-derived away, and + // namespaced OpenRouter ids parse. const staleOpenRouter = createModel({ id: "openai/gpt-5.6-terra", api: "openrouter", @@ -597,24 +596,24 @@ describe("model thinking derivation", () => { baseUrl: "https://openrouter.ai/api/v1", thinking: { mode: "effort", - efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, }, }); expect(staleOpenRouter.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "low", - low: "medium", - medium: "high", - high: "xhigh", - xhigh: "max", - }, + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], }); }); - it("keeps pre-5.6 and Devin-routed GPT models off the shifted effort map", () => { + it("keeps pre-5.6 and Devin-routed GPT models on their own effort surfaces", () => { const gpt55 = createModel({ id: "gpt-5.5", api: "openai-responses", @@ -628,7 +627,7 @@ describe("model thinking derivation", () => { expect(gpt55.thinking?.effortMap).toBeUndefined(); // Devin selects effort by routing to per-tier sibling model ids, never - // via a wire reasoning.effort field — the shifted map must not attach. + // via a wire reasoning.effort field — no effort map may attach. const devin = createModel({ id: "gpt-5-6-sol", api: "devin-agent", @@ -636,20 +635,20 @@ describe("model thinking derivation", () => { baseUrl: "https://server.codeium.com", thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], effortRouting: { off: "gpt-5-6-sol-none", - minimal: "gpt-5-6-sol-low", - low: "gpt-5-6-sol-medium", - medium: "gpt-5-6-sol-high", - high: "gpt-5-6-sol-xhigh", - xhigh: "gpt-5-6-sol-max", + low: "gpt-5-6-sol-low", + medium: "gpt-5-6-sol-medium", + high: "gpt-5-6-sol-high", + xhigh: "gpt-5-6-sol-xhigh", + max: "gpt-5-6-sol-max", }, }, }); expect(devin.thinking?.effortMap).toBeUndefined(); - expect(devin.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(devin.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); }); }); @@ -712,7 +711,7 @@ describe("model thinking runtime helpers", () => { ); }); - it("maps GLM-5.2 xhigh to Z.AI provider-native max", () => { + it("exposes the Z.AI GLM-5.2 high/max wire pair directly", () => { const model = createModel({ id: "glm-5.2", api: "openai-completions", @@ -723,19 +722,15 @@ describe("model thinking runtime helpers", () => { expect(model.thinking).toEqual({ mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], - effortMap: { - minimal: "none", - low: "high", - medium: "high", - high: "high", - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], }); - expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh); + expect(requireSupportedEffort(model, Effort.Max)).toBe(Effort.Max); + expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(/Supported efforts: high, max/); + // Selecting a retired tier clamps down instead of erroring in UI flows. + expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High); }); - it("maps Ollama Cloud GLM-5.2 xhigh to max and hides unsupported lower efforts", () => { + it("exposes Ollama Cloud GLM-5.2 high/max and hides unsupported lower efforts", () => { const model = createModel({ id: "glm-5.2", api: "ollama-chat", @@ -745,14 +740,11 @@ describe("model thinking runtime helpers", () => { expect(model.thinking).toEqual({ mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { - xhigh: "max", - }, + efforts: [Effort.High, Effort.Max], }); expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High); - expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh); - expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/Supported efforts: high, xhigh/); + expect(requireSupportedEffort(model, Effort.Max)).toBe(Effort.Max); + expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/Supported efforts: high, max/); }); it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => { @@ -774,7 +766,7 @@ describe("model thinking runtime helpers", () => { ); }); - it("exposes xhigh for OpenRouter-hosted Anthropic adaptive models", () => { + it("exposes wire-exact adaptive ladders for OpenRouter-hosted Anthropic models", () => { const fable = createModel({ id: "anthropic/claude-fable-5", api: "openai-completions", @@ -795,12 +787,13 @@ describe("model thinking runtime helpers", () => { api: "openai-completions", provider: "openrouter", }); - expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh); - expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh); + expect(fable.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus46.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High); - expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.XHigh); - expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh); + expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.Max); + expect(requireSupportedEffort(fable, Effort.Max)).toBe(Effort.Max); expect(requireSupportedEffort(sonnet5, Effort.XHigh)).toBe(Effort.XHigh); + expect(() => requireSupportedEffort(opus46, Effort.XHigh)).toThrow(/not supported/); }); it("enables xhigh for openai-responses and openai-codex-responses APIs", () => { diff --git a/packages/catalog/test/ollama-cloud-provider.test.ts b/packages/catalog/test/ollama-cloud-provider.test.ts index 4f640c145..4f85dac46 100644 --- a/packages/catalog/test/ollama-cloud-provider.test.ts +++ b/packages/catalog/test/ollama-cloud-provider.test.ts @@ -141,8 +141,7 @@ describe("ollama-cloud provider support", () => { expect(model?.reasoning).toBe(true); expect(built?.thinking).toEqual({ mode: "effort", - efforts: [Effort.High, Effort.XHigh], - effortMap: { xhigh: "max" }, + efforts: [Effort.High, Effort.Max], }); }); @@ -286,7 +285,7 @@ describe("ollama-cloud provider support", () => { expect(result.errorMessage).toContain("prompt filled the context window"); }); - test("sends max for GLM-5.2 xhigh reasoning on Ollama Cloud", async () => { + test("sends native max for GLM-5.2 max reasoning on Ollama Cloud", async () => { let requestBody: Record | undefined; const fetchMock: FetchImpl = vi.fn(async (_input, init) => { requestBody = JSON.parse(String(init?.body ?? "{}")) as Record; @@ -302,7 +301,7 @@ describe("ollama-cloud provider support", () => { provider: "ollama-cloud", baseUrl: "https://ollama.com", reasoning: true, - thinking: { mode: "effort", efforts: [Effort.High, Effort.XHigh], effortMap: { [Effort.XHigh]: "max" } }, + thinking: { mode: "effort", efforts: [Effort.High, Effort.Max] }, input: ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, @@ -315,7 +314,7 @@ describe("ollama-cloud provider support", () => { { apiKey: "cloud-test-key", fetch: fetchMock, - reasoning: Effort.XHigh, + reasoning: Effort.Max, }, ).result(); diff --git a/packages/catalog/test/ollama-provider.test.ts b/packages/catalog/test/ollama-provider.test.ts index acbf10fb3..aebd75789 100644 --- a/packages/catalog/test/ollama-provider.test.ts +++ b/packages/catalog/test/ollama-provider.test.ts @@ -3,6 +3,7 @@ import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama"; import type { Context, Tool } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking"; import { ollamaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; @@ -87,10 +88,14 @@ describe("ollama local provider discovery", () => { const builtReasoningModel = reasoningModel ? buildModel(reasoningModel) : undefined; const builtPlainModel = plainModel ? buildModel(plainModel) : undefined; - // Ollama's OpenAI-compatible endpoint rejects "minimal" with HTTP 400; - // reasoning models must bake a thinking effort map to an accepted level (low). + // Ollama's OpenAI-compatible endpoint accepts low/medium/high/max; + // reasoning models carry that wire-exact ladder with no remapping + // (minimal/xhigh never reach the wire because they are not offered). expect(reasoningModel?.reasoning).toBe(true); - expect(builtReasoningModel?.thinking?.effortMap).toMatchObject({ minimal: "low" }); + expect(builtReasoningModel?.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], + }); // Non-reasoning models never send an effort, so they carry no thinking metadata. expect(plainModel?.reasoning).toBe(false); expect(builtPlainModel?.thinking).toBeUndefined(); @@ -150,7 +155,7 @@ describe("ollama tool forcing", () => { }); }); -describe("ollama reasoning effort backfill (buildModel)", () => { +describe("ollama reasoning effort normalization (buildModel)", () => { const staleOllamaSpec = ( api: TApi, compat?: ModelSpec["compat"], @@ -170,38 +175,32 @@ describe("ollama reasoning effort backfill (buildModel)", () => { compat, }) as ModelSpec; - test("stamps the effort map on a stale ollama responses spec lacking compat", () => { - // A cache row or hand-written config written before the remap existed: - // reasoning-capable, `minimal` offered, but no reasoningEffortMap. The - // builder must backfill it so the wire never sends raw `minimal`/`xhigh`. + test("normalizes a stale ollama responses spec to the wire-exact ladder", () => { + // A cache row or hand-written config from the remap era: reasoning-capable + // with `minimal` offered. The builder must normalize the ladder so the + // wire never sends raw `minimal`/`xhigh`. const model = buildModel(staleOllamaSpec("openai-responses")); - expect(model.compat.reasoningEffortMap).toMatchObject({ minimal: "low", xhigh: "max" }); - // xhigh drops out of thinking.effortMap — it is not an offered effort. - expect(model.thinking?.effortMap).toEqual({ minimal: "low" }); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); + // Retired tiers clamp instead of erroring. + expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Low); + expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High); }); - test("backfills openai-completions ollama specs too", () => { + test("normalizes openai-completions ollama specs too", () => { const model = buildModel(staleOllamaSpec("openai-completions")); - expect(model.compat.reasoningEffortMap).toMatchObject({ minimal: "low", xhigh: "max" }); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.Max]); }); - test("explicit overrides win while missing ollama defaults stay", () => { - const model = buildModel(staleOllamaSpec("openai-responses", { reasoningEffortMap: { minimal: "medium" } })); - expect(model.compat.reasoningEffortMap).toEqual({ minimal: "medium", xhigh: "max" }); + test("explicit compat overrides survive for live tiers", () => { + const model = buildModel(staleOllamaSpec("openai-responses", { reasoningEffortMap: { high: "medium" } })); + expect(model.compat.reasoningEffortMap).toEqual({ high: "medium" }); + expect(model.thinking?.effortMap).toEqual({ high: "medium" }); }); test("leaves non-ollama providers untouched", () => { const model = buildModel({ ...staleOllamaSpec("openai-responses"), provider: "custom" }); expect(model.compat.reasoningEffortMap).toEqual({}); - }); - - test("merges the ollama defaults into the whenThinking variant", () => { - const model = buildModel( - staleOllamaSpec("openai-completions", { whenThinking: { reasoningEffortMap: { minimal: "medium" } } }), - ); - // The thinking-engaged variant must keep the xhigh default; otherwise a - // partial whenThinking override would re-leak raw `minimal`/`xhigh`. - expect(model.compat.whenThinking?.reasoningEffortMap).toEqual({ minimal: "medium", xhigh: "max" }); - expect(model.compat.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "max" }); + expect(model.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); }); }); diff --git a/packages/catalog/test/sakana-provider.test.ts b/packages/catalog/test/sakana-provider.test.ts index f7b9d3ac2..0456be443 100644 --- a/packages/catalog/test/sakana-provider.test.ts +++ b/packages/catalog/test/sakana-provider.test.ts @@ -60,8 +60,8 @@ describe("Sakana AI provider support", () => { expect(bundled.find(model => model.id === "fugu-ultra-20260615")?.contextWindow).toBe(1_000_000); for (const model of bundled) { expect(model.api).toBe("openai-responses"); - expect(model.thinking?.efforts).toEqual([Effort.High, Effort.XHigh]); - expect(model.thinking?.effortMap?.[Effort.XHigh]).toBe("max"); + expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.effortMap).toBeUndefined(); expect((model.compat as ResolvedOpenAIResponsesCompat).includeEncryptedReasoning).toBe(false); expect((model.compat as ResolvedOpenAIResponsesCompat).streamIdleTimeoutMs).toBe(0); } @@ -95,8 +95,8 @@ describe("Sakana AI provider support", () => { expect(models?.map(model => model.id)).toEqual(["fugu", "fugu-next", "fugu-ultra"]); const fuguNext = models?.find(model => model.id === "fugu-next"); expect(fuguNext?.reasoning).toBe(true); - expect(fuguNext?.thinking?.efforts).toEqual([Effort.High, Effort.XHigh]); - expect(fuguNext?.thinking?.effortMap?.[Effort.XHigh]).toBe("max"); + expect(fuguNext?.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(fuguNext?.thinking?.effortMap).toBeUndefined(); expect(fuguNext?.compat?.includeEncryptedReasoning).toBe(false); }); diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index a128d59f8..a641e6ec9 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -118,12 +118,11 @@ describe("umans provider catalog", () => { thinking: { mode: "anthropic-budget-effort", defaultLevel: "high", - efforts: ["high", "xhigh"], - effortMap: { xhigh: "max" }, + efforts: ["high", "max"], }, }); if (!glm52) throw new Error("Umans GLM 5.2 was not discovered"); - expect(glm52.thinking?.effortMap).toEqual({ [Effort.XHigh]: "max" }); + expect(glm52.thinking?.effortMap).toBeUndefined(); expect(glm52.thinking?.defaultLevel).toBe(Effort.High); }); @@ -307,15 +306,15 @@ describe("umans provider catalog", () => { }); }); - it("bundles Umans GLM 5.2 high/max reasoning metadata with the max wire effort", () => { + it("bundles Umans GLM 5.2 with the wire-exact high/max ladder", () => { const providers = modelsJson as Record>; const model = providers.umans?.["umans-glm-5.2"]; expect(model).toBeDefined(); expect(model.thinking).toMatchObject({ mode: "anthropic-budget-effort", - efforts: ["high", "xhigh"], - effortMap: { xhigh: "max" }, + efforts: ["high", "max"], }); + expect(model.thinking?.effortMap).toBeUndefined(); }); }); diff --git a/packages/catalog/test/variant-collapse.test.ts b/packages/catalog/test/variant-collapse.test.ts index ae373b6bf..cbfc93781 100644 --- a/packages/catalog/test/variant-collapse.test.ts +++ b/packages/catalog/test/variant-collapse.test.ts @@ -17,6 +17,7 @@ import { ANTIGRAVITY_VARIANT_COLLAPSE_TABLE, collapseEffortVariants, collapseEffortVariantsAcrossProviders, + DEVIN_VARIANT_COLLAPSE_TABLE, deriveThinkingPairFamilies, GEMINI_CLI_VARIANT_COLLAPSE_TABLE, getVariantAliasSources, @@ -526,6 +527,45 @@ describe("collapseEffortVariantsAcrossProviders", () => { }); }); +describe("Devin tier routing", () => { + const family = (id: string) => { + const found = DEVIN_VARIANT_COLLAPSE_TABLE.families.find(f => f.id === id); + if (!found) throw new Error(`Devin family ${id} missing`); + return found; + }; + + it("routes user efforts 1:1 onto per-tier siblings including max", () => { + const opus = family("claude-opus-4-8"); + expect(opus.routing).toEqual({ + [Effort.Low]: "claude-opus-4-8-low", + [Effort.Medium]: "claude-opus-4-8-medium", + [Effort.High]: "claude-opus-4-8-high", + [Effort.XHigh]: "claude-opus-4-8-xhigh", + [Effort.Max]: "claude-opus-4-8-max", + }); + expect(opus.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]); + expect(opus.thinking.requiresEffort).toBe(true); + + const sol = family("gpt-5-6-sol"); + expect(sol.routing[Effort.Max]).toBe("gpt-5-6-sol-max"); + expect(sol.routing[Effort.Low]).toBe("gpt-5-6-sol-low"); + expect(sol.routing.off).toBe("gpt-5-6-sol-none"); + expect(sol.routing[Effort.Minimal]).toBeUndefined(); + }); + + it("keeps families without a -max sibling on the xhigh ceiling", () => { + const solFast = family("gpt-5-6-sol-fast"); + expect(solFast.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(solFast.routing[Effort.Max]).toBeUndefined(); + expect(solFast.routing[Effort.XHigh]).toBe("gpt-5-6-sol-xhigh-priority"); + + const gpt55 = family("gpt-5-5"); + expect(gpt55.thinking.efforts).toEqual([Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + expect(gpt55.routing[Effort.Minimal]).toBeUndefined(); + expect(gpt55.routing[Effort.Max]).toBeUndefined(); + }); +}); + describe("variant aliases", () => { it("resolves members and recycled ids per provider", () => { expect(resolveVariantAlias("google-antigravity", "gemini-3.5-flash-low")).toBe("gemini-3.5-flash"); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 658e3a904..214127a96 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,10 +6,18 @@ - Renamed the bundled agent `explore` to `scout` (including renaming its configuration keys, prompt files, and task definitions). Any configurations, allowlists, or invocations referencing `explore` must now use `scout`. +### Added + +- Added `max` as a native, first-class thinking tier for supported models +- Added `thinkingBudgets.max` configuration setting +- Updated terminal theme to support optional `thinkingMax` border color and icons + +- Added a real `max` thinking level above `xhigh` with an optional `thinkingMax` theme border color (falls back to `thinkingXhigh`). `max` owns the top status-line icons (`◉` unicode, fire nerd-font, `[max]` ascii); the nerd-font preset uses an empty-to-full battery ramp for `minimal` through `xhigh` and shuffle while automatic effort is unresolved. `max` appears in cycling, selectors, `--thinking`, `:max` model suffixes, role scopes, settings, and completions on models that genuinely support it; on other models it clamps down like any unsupported tier. + ### Changed - Renamed the bundled agent `explore` to `scout` (including all internal references, prompt definitions, and task tool configurations). Any custom configurations or task invocations referencing `explore` must now use `scout`. - +- Changed `max` from a parse-time alias of `xhigh` to a distinct level everywhere (CLI flag, `:max` suffix, `defaultThinkingLevel`, ACP/RPC): the effort a model receives is now exactly the tier its wire supports, with no shifted remapping. Ultrathink now requests `max` (clamped per model); automatic thinking still tops out at `xhigh`. Added `thinkingBudgets.max` (default 32768). - Fixed collapsed compacted session transcript rebuilds reattaching snapcompact archive image frames to the live TUI, avoiding large retained JSC heaps on resume and transcript refresh. ([#4979](https://github.com/can1357/oh-my-pi/issues/4979)) ### Fixed diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 82c4e0545..7fb1491b6 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -690,7 +690,7 @@ function normalizeSuppressedSelector( const trimmed = selector.trim(); if (!trimmed) return trimmed; const parsed = parseModelString(trimmed, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => hasLiveModel?.(provider, id) === true, }); diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index f09310b51..32ed6116f 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -79,23 +79,24 @@ export interface ScopedModel { } interface ThinkingSuffixOptions { - allowMaxAlias?: boolean; + allowMaxSuffix?: boolean; allowAutoAlias?: boolean; } interface ModelStringParseOptions extends ThinkingSuffixOptions { isLiteralModelId?: (provider: string, id: string) => boolean; } -// Alias-suffix recognition for the model-pattern parser: `:max` maps to xhigh -// and `:auto` maps to the auto sentinel. Both are gated behind the alias flags -// (and the literal-id / exact-match guards on the callers) so a real model id -// ending in `:max` / `:auto` isn't silently reinterpreted as a thinking suffix. -const MAX_THINKING_SUFFIX_OPTIONS: ThinkingSuffixOptions = { allowMaxAlias: true, allowAutoAlias: true }; +// Suffix recognition for the model-pattern parser: `:max` is a real thinking +// level and `:auto` maps to the auto sentinel. Both are gated behind flags +// (and the literal-id / exact-match guards on the callers) because real model +// ids end in `:max` (e.g. `glm-4.7:max`) — an ungated split would silently +// reinterpret them as a thinking suffix. +const MAX_THINKING_SUFFIX_OPTIONS: ThinkingSuffixOptions = { allowMaxSuffix: true, allowAutoAlias: true }; function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): ConfiguredThinkingLevel | undefined { const level = parseThinkingLevel(value); + if (level === ThinkingLevel.Max) return options?.allowMaxSuffix === true ? level : undefined; if (level !== undefined) return level; - if (options?.allowMaxAlias === true && value === "max") return ThinkingLevel.XHigh; if (options?.allowAutoAlias === true && value === AUTO_THINKING) return AUTO_THINKING; return undefined; } @@ -104,8 +105,9 @@ function parseThinkingSuffix(value: string, options?: ThinkingSuffixOptions): Co * Split a trailing `:` thinking selector off a model pattern. * * `level` is set when the suffix parses as a concrete thinking level (or, when - * the caller opts in via `allowMaxAlias`/`allowAutoAlias`, the `:max` / `:auto` - * aliases); `base` then has the suffix stripped. Otherwise `base` is the input. + * the caller opts in via `allowMaxSuffix`/`allowAutoAlias`, the guarded `:max` + * level / `:auto` sentinel); `base` then has the suffix stripped. Otherwise + * `base` is the input. * `minColonIndex` requires the colon to appear strictly after that index — * role-alias callers pass `PREFIX_MODEL_ROLE.length` so the base is at least * as long as the `pi/` prefix. @@ -183,7 +185,7 @@ export function parseModelString( // Strip strict thinking level suffixes first (e.g. "claude-sonnet-4-6:high" -> id "claude-sonnet-4-6", thinkingLevel "high"). const strict = splitThinkingSuffix(id); if (strict.level) return { provider, id: strict.base, thinkingLevel: strict.level }; - // `max` is a provider-facing alias for xhigh, but real model IDs can end in + // `max` is a real thinking level, but real model IDs can also end in // `:max`. Context-aware callers pass a literal lookup so those models win. const maxAlias = splitThinkingSuffix(id, -1, options); if (maxAlias.level) { @@ -239,7 +241,7 @@ function getOpenRouterRouteSuffix(modelId: string): { baseId: string; suffix: st } const suffix = modelId.slice(colonIdx + 1).trim(); - // `max` is a thinking-level alias (xhigh), never an OpenRouter route suffix, so + // `max` is a thinking-level suffix, never an OpenRouter route suffix, so // `openrouter/:max` falls through to the max-aware selector split instead of // being cloned into a literal `:max` model id with the reasoning level lost. if (!suffix || parseThinkingSuffix(suffix, MAX_THINKING_SUFFIX_OPTIONS)) { @@ -766,7 +768,7 @@ function parseModelPatternWithContext( // No match - try stripping a valid thinking suffix and recursing. // `max` is accepted only after the full pattern failed, so literal model IDs - // ending in `:max` keep winning over the alias. + // ending in `:max` keep winning over the thinking suffix. const { base, level } = splitThinkingSuffix(pattern, -1, MAX_THINKING_SUFFIX_OPTIONS); if (level) { const result = parseModelPatternWithContext(base, availableModels, context, options); diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 7d886bff5..1309ad8de 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -23,6 +23,7 @@ const ReasoningEffortMapSchema = type({ "medium?": "string", "high?": "string", "xhigh?": "string", + "max?": "string", }); const OpenAICompatFields = { @@ -74,13 +75,13 @@ const ApiSchema = type( '"openai-completions" | "openai-responses" | "openai-codex-responses" | "azure-openai-responses" | "anthropic-messages" | "google-generative-ai" | "google-gemini-cli" | "google-vertex"', ); -const EffortSchema = type('"minimal" | "low" | "medium" | "high" | "xhigh"'); +const EffortSchema = type('"minimal" | "low" | "medium" | "high" | "xhigh" | "max"'); const ThinkingControlModeSchema = type( '"effort" | "budget" | "google-level" | "anthropic-adaptive" | "anthropic-budget-effort"', ); -const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh"] as const; +const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh", "max"] as const; /** * Accepts the canonical `efforts` vocabulary plus the legacy diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 5ee3269d1..81f4ef7dd 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -936,7 +936,7 @@ export const SETTINGS_SCHEMA = { // Reasoning and prompts defaultThinkingLevel: { type: "enum", - values: [...THINKING_EFFORTS, AUTO_THINKING, "max"], + values: [...THINKING_EFFORTS, AUTO_THINKING], default: "high", ui: { tab: "model", @@ -4964,6 +4964,8 @@ export const SETTINGS_SCHEMA = { "thinkingBudgets.high": { type: "number", default: 16384 }, "thinkingBudgets.xhigh": { type: "number", default: 32768 }, + + "thinkingBudgets.max": { type: "number", default: 32768 }, } as const; // ═══════════════════════════════════════════════════════════════════════════ @@ -5180,6 +5182,7 @@ export interface ThinkingBudgetsSettings { medium: number; high: number; xhigh: number; + max: number; } export interface SttSettings { diff --git a/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json b/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json index 69c56126c..80aa3e6cf 100644 --- a/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json +++ b/packages/coding-agent/src/modes/theme/defaults/dark-poimandres.json @@ -122,11 +122,12 @@ "status.pending": "◌", "nav.cursor": "▸", "nav.selected": "▴", - "thinking.minimal": "◌", - "thinking.low": "◍", - "thinking.medium": "◎", - "thinking.high": "◉", - "thinking.xhigh": "●", + "thinking.minimal": "∘", + "thinking.low": "◌", + "thinking.medium": "◍", + "thinking.high": "◎", + "thinking.xhigh": "◉", + "thinking.max": "●", "icon.model": "◇", "icon.plan": "◈", "icon.goal": "⊙", diff --git a/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json b/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json index 8b00ca22e..a6da984a9 100644 --- a/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json +++ b/packages/coding-agent/src/modes/theme/defaults/light-poimandres.json @@ -122,11 +122,12 @@ "status.pending": "◌", "nav.cursor": "▸", "nav.selected": "▴", - "thinking.minimal": "◌", - "thinking.low": "◍", - "thinking.medium": "◎", - "thinking.high": "◉", - "thinking.xhigh": "●", + "thinking.minimal": "∘", + "thinking.low": "◌", + "thinking.medium": "◍", + "thinking.high": "◎", + "thinking.xhigh": "◉", + "thinking.max": "●", "icon.model": "◇", "icon.plan": "◈", "icon.goal": "⊙", diff --git a/packages/coding-agent/src/modes/theme/theme-schema.json b/packages/coding-agent/src/modes/theme/theme-schema.json index 3fd9972b4..8dd7eb41c 100644 --- a/packages/coding-agent/src/modes/theme/theme-schema.json +++ b/packages/coding-agent/src/modes/theme/theme-schema.json @@ -33,7 +33,7 @@ }, "colors": { "type": "object", - "description": "Theme color definitions (all required)", + "description": "Theme color definitions (all required except thinkingMax)", "required": [ "accent", "border", @@ -301,7 +301,11 @@ }, "thinkingXhigh": { "$ref": "#/$defs/colorValue", - "description": "Thinking level border: xhigh (OpenAI codex-max only)" + "description": "Thinking level border: xhigh" + }, + "thinkingMax": { + "$ref": "#/$defs/colorValue", + "description": "Thinking level border: max (optional; falls back to thinkingXhigh when omitted)" }, "bashMode": { "$ref": "#/$defs/colorValue", diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index 77140221b..ca0a6397f 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -142,6 +142,7 @@ export type SymbolKey = | "thinking.medium" | "thinking.high" | "thinking.xhigh" + | "thinking.max" | "thinking.autoPending" // Checkboxes | "checkbox.checked" @@ -344,11 +345,12 @@ const UNICODE_SYMBOLS: SymbolMap = { // Compaction divider "icon.camera": "📷", // Thinking levels - "thinking.minimal": "◔ min", - "thinking.low": "◑ low", - "thinking.medium": "◒ med", - "thinking.high": "◕ high", - "thinking.xhigh": "◉ xhigh", + "thinking.minimal": "○ min", + "thinking.low": "◔ low", + "thinking.medium": "◑ med", + "thinking.high": "◒ high", + "thinking.xhigh": "◕ xhigh", + "thinking.max": "◉ max", "thinking.autoPending": "⟳", // Checkboxes "checkbox.checked": "☑", @@ -639,19 +641,15 @@ const NERD_SYMBOLS: SymbolMap = { "icon.mic": "\uf130", // Compaction divider - fa-camera-retro "icon.camera": "\uf083", - // Thinking Levels - emoji labels - // pick: 🤨 min | alt:  min  min - "thinking.minimal": "\u{F0E7} min", - // pick: 🤔 low | alt:  low  low - "thinking.low": "\u{F10C} low", - // pick: 🤓 med | alt:  med  med - "thinking.medium": "\u{F192} med", - // pick: 🤯 high | alt:  high  high - "thinking.high": "\u{F111} high", - // pick: 🧠 xhi | alt:  xhi  xhi - "thinking.xhigh": "\u{F06D} xhi", - // pick: (fa-circle-o-notch) | alt: 󰂼 (nf-md-cached) ⟳ - "thinking.autoPending": "\uf1ce", + // Thinking levels — empty-to-full battery ramp, then fire. + "thinking.minimal": "\u{F244} min", + "thinking.low": "\u{F243} low", + "thinking.medium": "\u{F242} med", + "thinking.high": "\u{F241} high", + "thinking.xhigh": "\u{F240} xhi", + "thinking.max": "\u{F06D} max", + // Auto mode uses shuffle until the model resolves its thinking level. + "thinking.autoPending": "\u{F074}", // Checkboxes // pick:  | alt:   "checkbox.checked": "\uf14a", @@ -868,6 +866,7 @@ const ASCII_SYMBOLS: SymbolMap = { "thinking.medium": "[med]", "thinking.high": "[high]", "thinking.xhigh": "[xhi]", + "thinking.max": "[max]", "thinking.autoPending": "[~]", // Checkboxes "checkbox.checked": "[x]", @@ -1059,6 +1058,7 @@ const themeColorsSchema = type({ thinkingMedium: "string | number", thinkingHigh: "string | number", thinkingXhigh: "string | number", + "thinkingMax?": "string | number", bashMode: "string | number", pythonMode: "string | number", statusLineBg: "string | number", @@ -1163,6 +1163,7 @@ export type ThemeColor = | "thinkingMedium" | "thinkingHigh" | "thinkingXhigh" + | "thinkingMax" | "bashMode" | "pythonMode" | "statusLineSep" @@ -1225,6 +1226,7 @@ const THEME_COLOR_RECORD = { thinkingMedium: true, thinkingHigh: true, thinkingXhigh: true, + thinkingMax: true, bashMode: true, pythonMode: true, statusLineSep: true, @@ -1675,6 +1677,9 @@ export class Theme { return (str: string) => this.fg("thinkingHigh", str); case "xhigh": return (str: string) => this.fg("thinkingXhigh", str); + case "max": + // thinkingMax is optional; themes without it resolve to the xhigh color. + return (str: string) => this.fg(this.#fgColors.thinkingMax ? "thinkingMax" : "thinkingXhigh", str); default: return (str: string) => this.fg("thinkingOff", str); } @@ -1861,6 +1866,7 @@ export class Theme { medium: this.#symbols["thinking.medium"], high: this.#symbols["thinking.high"], xhigh: this.#symbols["thinking.xhigh"], + max: this.#symbols["thinking.max"], autoPending: this.#symbols["thinking.autoPending"], }; } diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 16a066675..8afd72517 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1309,7 +1309,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} for (let i = 0; i < sessionModelStrings.length; i++) { const sessionModelStr = sessionModelStrings[i]; const parsedModel = parseModelString(sessionModelStr, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, }); @@ -1953,7 +1953,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} for (let i = 0; i < sessionRetryLimit; i++) { const sessionModelStr = sessionModelStrings[i]; const parsedModel = parseModelString(sessionModelStr, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelRegistry.find(provider, id) !== undefined, }); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index a98f48521..e6702ad36 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1081,7 +1081,7 @@ function parseRetryFallbackSelector( const trimmed = selector.trim(); if (!trimmed) return undefined; const parsed = parseModelString(trimmed, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => modelLookup?.find(provider, id) !== undefined, }); @@ -9260,7 +9260,7 @@ export class AgentSession { } /** - * Cycle to next thinking level: off → auto → minimal..xhigh → off. + * Cycle to next thinking level: off → auto → minimal..max → off. * @returns New selector, or undefined if model doesn't support thinking */ cycleThinkingLevel(): ConfiguredThinkingLevel | undefined { @@ -9302,8 +9302,9 @@ export class AgentSession { let resolved: Effort | undefined; if (this.#magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) { // The user explicitly asked for maximum thinking; bypass the classifier - // and jump straight to the highest auto-supported level for this model. - resolved = clampAutoThinkingEffort(model, Effort.XHigh); + // (and its xhigh auto ceiling) and jump straight to the highest + // supported level for this model. + resolved = clampAutoThinkingEffort(model, Effort.Max); } else { const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), AgentSession.#AUTO_THINKING_TIMEOUT_MS); @@ -11967,7 +11968,7 @@ export class AgentSession { if (!trimmedTarget) return undefined; const parsed = parseModelString(trimmedTarget, { - allowMaxAlias: true, + allowMaxSuffix: true, allowAutoAlias: true, isLiteralModelId: (provider, id) => availableModels.some(model => model.provider === provider && model.id === id), diff --git a/packages/coding-agent/src/system-prompt.test.ts b/packages/coding-agent/src/system-prompt.test.ts index 673085068..2f93dbbc4 100644 --- a/packages/coding-agent/src/system-prompt.test.ts +++ b/packages/coding-agent/src/system-prompt.test.ts @@ -123,29 +123,36 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { }, 15_000); it("kills the GPU probe at the prep deadline", async () => { - const result = await runProbeScenario({ runs: 1, sleepSeconds: 7, holdStdoutOpen: true }); + const result = await runProbeScenario({ runs: 1, sleepSeconds: 12, holdStdoutOpen: true }); expect(result.cached).toEqual({ gpu: null }); + // Probe is SIGKILLed at ~4.5s and the drain wait is bounded, so in-child + // time sits near the deadline; waiting on the descendant would push it + // past the 12s sleep. expect(result.elapsedMs).toBeLessThan(6500); - // Codex#3838: the child process MUST exit shortly after the deadline, - // not linger until a descendant holding stdout (sleep 7) exits on its own. - expect(result.childElapsedMs).toBeLessThan(6500); - }, 15_000); + // Codex#3838: the child process MUST exit shortly after the deadline, not + // linger until a descendant holding stdout (sleep 12) exits on its own. + // The bound over in-child time budgets bun spawn/startup on loaded runners + // while staying far below the descendant's 12s exit. + expect(result.childElapsedMs).toBeLessThan(9000); + }, 20_000); it("does not wait on stdout held by a descendant after a successful probe", async () => { - const result = await runProbeScenario({ runs: 1, sleepSeconds: 3, descendantHoldsStdout: true }); + const result = await runProbeScenario({ runs: 1, sleepSeconds: 8, descendantHoldsStdout: true }); expect(result.cached).toEqual({ gpu: null }); // Probe exits 0 immediately but leaves a backgrounded sleep holding the stdout // pipe. The success path MUST bound the drain wait, not block until sleep exits. expect(result.elapsedMs).toBeLessThan(2000); - expect(result.childElapsedMs).toBeLessThan(2000); - }, 15_000); + // Budgets bun spawn/startup overhead; blocking on the descendant would + // take at least the 8s sleep. + expect(result.childElapsedMs).toBeLessThan(5000); + }, 20_000); it("keeps probe output captured before a descendant delays EOF", async () => { const result = await runProbeScenario({ runs: 1, - sleepSeconds: 3, + sleepSeconds: 8, descendantHoldsStdout: true, validOutput: "00:02.0 VGA compatible controller: NVIDIA TestGPU", }); @@ -154,8 +161,10 @@ describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => { // Captured stdout MUST be cached, not discarded as if the probe failed. expect(result.cached).toEqual({ gpu: "02.0 VGA compatible controller: NVIDIA TestGPU" }); expect(result.elapsedMs).toBeLessThan(2000); - expect(result.childElapsedMs).toBeLessThan(2000); - }, 15_000); + // Budgets bun spawn/startup overhead; blocking on the descendant would + // take at least the 8s sleep. + expect(result.childElapsedMs).toBeLessThan(5000); + }, 20_000); }); describe.skipIf(process.platform !== "linux")("system prompt CPU model", () => { diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 7a6e83a8b..0bf10760c 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -33,7 +33,12 @@ const THINKING_LEVEL_METADATA: Record = { [ThinkingLevel.XHigh]: { value: ThinkingLevel.XHigh, label: "xhigh", - description: "Maximum reasoning (~32k tokens)", + description: "Extended reasoning (~32k tokens)", + }, + [ThinkingLevel.Max]: { + value: ThinkingLevel.Max, + label: "max", + description: "Maximum reasoning the model supports", }, }; @@ -43,7 +48,7 @@ const EFFORT_BY_SELECTOR: Readonly> = { [Effort.Medium]: Effort.Medium, [Effort.High]: Effort.High, [Effort.XHigh]: Effort.XHigh, - max: Effort.XHigh, + [Effort.Max]: Effort.Max, }; const THINKING_LEVEL_BY_SELECTOR: Readonly> = { [ThinkingLevel.Inherit]: ThinkingLevel.Inherit, @@ -53,6 +58,7 @@ const THINKING_LEVEL_BY_SELECTOR: Readonly> = { [ThinkingLevel.Medium]: ThinkingLevel.Medium, [ThinkingLevel.High]: ThinkingLevel.High, [ThinkingLevel.XHigh]: ThinkingLevel.XHigh, + [ThinkingLevel.Max]: ThinkingLevel.Max, }; function getOwnSelector(selectors: Readonly>, value: string | null | undefined): T | undefined { @@ -149,7 +155,6 @@ const AUTO_THINKING_METADATA: ConfiguredThinkingLevelMetadata = { */ export function parseConfiguredThinkingLevel(value: string | null | undefined): ConfiguredThinkingLevel | undefined { if (value === AUTO_THINKING) return AUTO_THINKING; - if (value === "max") return ThinkingLevel.XHigh; return parseThinkingLevel(value); } @@ -160,7 +165,7 @@ export function getConfiguredThinkingLevelMetadata(level: ConfiguredThinkingLeve /** * Thinking selectors accepted by the `--thinking` CLI flag, in display order: - * `off`, every concrete effort (`minimal`..`xhigh`), then `auto`. Single source + * `off`, every concrete effort (`minimal`..`max`), then `auto`. Single source * for the flag's `options` list, shell completions, and the "invalid level" * warning so all three stay in sync. */ @@ -168,7 +173,7 @@ export const CLI_THINKING_LEVELS: readonly string[] = [ThinkingLevel.Off, ...THI /** * Parses a `--thinking` CLI value. Accepts every {@link parseConfiguredThinkingLevel} - * selector (`off`, `auto`, `minimal`..`xhigh`, plus the `max` alias) but rejects + * selector (`off`, `auto`, `minimal`..`max`) but rejects * `inherit`: an explicit `inherit` on the command line would suppress the * settings/scoped-model fallback during startup resolution only to resolve back * to the provider default, which is never what the user means. @@ -211,9 +216,13 @@ export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort /** * The provisional concrete level shown while `auto` is configured but before a * turn has been classified. Prefers the model's `defaultLevel`, otherwise High, - * clamped into the auto range. Returns `undefined` for non-reasoning models. + * clamped into the auto range. Auto never provisions {@link Effort.Max} (the + * classifier ceiling is XHigh; only an explicit user request reaches Max), so a + * `defaultLevel` of `max` is capped at XHigh before clamping. Returns + * `undefined` for non-reasoning models. */ export function resolveProvisionalAutoLevel(model: Model | undefined): Effort | undefined { if (!model?.reasoning) return undefined; - return clampAutoThinkingEffort(model, model.thinking?.defaultLevel ?? Effort.High); + const preferred = model.thinking?.defaultLevel ?? Effort.High; + return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred); } diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 345d42df3..25c8c6791 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -8,7 +8,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; -import { parseModelPattern } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { parseModelPattern, parseModelString } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -1602,6 +1602,12 @@ describe("AgentSession retry fallback", () => { expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o")).toBe(true); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:low")).toBe(true); + // `:max` is a real thinking level now, not an xhigh alias — the two parse + // to distinct selectors... + expect(parseModelString("openai/gpt-4o:max", { allowMaxSuffix: true })?.thinkingLevel).toBe(Effort.Max); + expect(parseModelString("openai/gpt-4o:xhigh")?.thinkingLevel).toBe(Effort.XHigh); + // ...but suppression normalizes every thinking suffix to the base selector, + // so suppressing either still covers both. modelRegistry.suppressSelector("openai/gpt-4o:max", future); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:xhigh")).toBe(true); expect(modelRegistry.isSelectorSuppressed("openai/gpt-4o:max")).toBe(true); diff --git a/packages/coding-agent/test/agent-session-role-thinking.test.ts b/packages/coding-agent/test/agent-session-role-thinking.test.ts index a489a02b6..910908325 100644 --- a/packages/coding-agent/test/agent-session-role-thinking.test.ts +++ b/packages/coding-agent/test/agent-session-role-thinking.test.ts @@ -147,14 +147,16 @@ describe("AgentSession role model thinking behavior", () => { expect(toSlow?.thinkingLevel).toBe(Effort.High); expect(session.thinkingLevel).toBe(Effort.High); - session.setThinkingLevel(Effort.Minimal); - expect(session.thinkingLevel).toBe(Effort.Minimal); + // `medium` is supported on both ladders (4-6 dropped `minimal`), so the + // selection survives the role switch unclamped. + session.setThinkingLevel(Effort.Medium); + expect(session.thinkingLevel).toBe(Effort.Medium); const toDefault = await session.cycleRoleModels(["default", "slow"]); expect(toDefault?.role).toBe("default"); expect(toDefault?.model.id).toBe(defaultModel.id); - expect(toDefault?.thinkingLevel).toBe(Effort.Minimal); - expect(session.thinkingLevel).toBe(Effort.Minimal); + expect(toDefault?.thinkingLevel).toBe(Effort.Medium); + expect(session.thinkingLevel).toBe(Effort.Medium); }); it("applies slow role thinking even when plan shares the same model", async () => { @@ -231,6 +233,36 @@ describe("AgentSession role model thinking behavior", () => { expect(session.getAvailableThinkingLevels()).not.toContain("xhigh"); }); + it("clamps max selections down to the ladder ceiling on models without a max tier", async () => { + // Budget-mode sonnet-4-5 tops out at xhigh; a max request must clamp down. + const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + thinkingLevel: undefined, + }, + }); + const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-max-clamp.db")); + authStorages.push(authStorage); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-max-clamp.yml")); + + sessionSettings = Settings.isolated(); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: sessionSettings, + modelRegistry, + }); + + session.setThinkingLevel(Effort.Max); + expect(session.thinkingLevel).toBe(Effort.XHigh); + expect(session.getAvailableThinkingLevels()).not.toContain("max"); + }); + it("cycles through off and auto before returning to effort levels", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); @@ -267,6 +299,40 @@ describe("AgentSession role model thinking behavior", () => { expect(session.thinkingLevel).toBe(Effort.Minimal); }); + it("cycles through max as the final tier on a max-capable model", async () => { + const model = getAnthropicModelOrThrow("claude-opus-4-7"); + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + thinkingLevel: Effort.XHigh, + }, + }); + const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-cycle-max.db")); + authStorages.push(authStorage); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-cycle-max.yml")); + + sessionSettings = Settings.isolated(); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: sessionSettings, + modelRegistry, + }); + + const available = session.getAvailableThinkingLevels(); + expect(available.at(-1)).toBe(Effort.Max); + + session.setThinkingLevel(Effort.XHigh); + expect(session.cycleThinkingLevel()).toBe(Effort.Max); + expect(session.thinkingLevel).toBe(Effort.Max); + // max is the last tier: the wheel wraps back to off. + expect(session.cycleThinkingLevel()).toBe("off"); + }); + it("keeps auto configured while applying the classifier result as the effective level", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); await createSession({ @@ -472,7 +538,7 @@ describe("AgentSession role model thinking behavior", () => { expect(session.autoResolvedThinkingLevel()).toBeUndefined(); }); - it("maps ultrathink prompts directly to the highest auto-supported level", async () => { + it("maps ultrathink prompts to the model's highest supported level, clamped below max", async () => { const model = getAnthropicModelOrThrow("claude-sonnet-4-5"); await createSession({ initialModelId: model.id, @@ -483,7 +549,9 @@ describe("AgentSession role model thinking behavior", () => { const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low); session.setThinkingLevel(AUTO_THINKING); - const expected = clampAutoThinkingEffort(model, Effort.XHigh); + // sonnet-4-5 has no max tier, so the ultrathink jump clamps to xhigh. + const expected = clampAutoThinkingEffort(model, Effort.Max); + expect(expected).toBe(Effort.XHigh); await session.prompt("ultrathink through the unsafe refactor"); expect(classifierSpy).not.toHaveBeenCalled(); @@ -491,6 +559,24 @@ describe("AgentSession role model thinking behavior", () => { expect(session.autoResolvedThinkingLevel()).toBe(expected); }); + it("resolves ultrathink to max on max-capable models", async () => { + const model = getAnthropicModelOrThrow("claude-opus-4-7"); + await createSession({ + initialModelId: model.id, + initialThinkingLevel: Effort.High, + modelRoles: { default: `${model.provider}/${model.id}` }, + }); + vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low); + + session.setThinkingLevel(AUTO_THINKING); + await session.prompt("ultrathink through the unsafe refactor"); + + expect(classifierSpy).not.toHaveBeenCalled(); + expect(session.thinkingLevel).toBe(Effort.Max); + expect(session.autoResolvedThinkingLevel()).toBe(Effort.Max); + }); + it("keeps auto effectively off for non-reasoning models", async () => { const model = getBundledModel("openai", "gpt-4o-mini"); if (!model) throw new Error("Expected bundled gpt-4o-mini model"); diff --git a/packages/coding-agent/test/auto-thinking-classifier.test.ts b/packages/coding-agent/test/auto-thinking-classifier.test.ts index 912b1e6d8..29f67b392 100644 --- a/packages/coding-agent/test/auto-thinking-classifier.test.ts +++ b/packages/coding-agent/test/auto-thinking-classifier.test.ts @@ -3,6 +3,7 @@ import * as path from "node:path"; import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import * as ai from "@oh-my-pi/pi-ai"; import { Effort, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { classifyDifficulty, @@ -71,7 +72,7 @@ describe("auto thinking classifier helpers", () => { it("parses CLI --thinking selectors while rejecting inherit", () => { expect(parseCliThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off); expect(parseCliThinkingLevel(AUTO_THINKING)).toBe(AUTO_THINKING); - expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.XHigh); + expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.Max); expect(parseCliThinkingLevel(ThinkingLevel.Inherit)).toBeUndefined(); expect(parseCliThinkingLevel("bogus")).toBeUndefined(); }); @@ -182,6 +183,24 @@ describe("auto thinking classifier helpers", () => { expect(clampAutoThinkingEffort(model, Effort.Minimal)).toBe(Effort.Low); }); + it("clamps max down to the ladder ceiling on models without a max tier", () => { + const xhighCeilingModel = buildModel({ + id: "mock-xhigh-ceiling", + name: "Mock XHigh Ceiling", + api: "openai-completions", + provider: "mock", + baseUrl: "https://example.com", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 4096, + }); + + expect(clampAutoThinkingEffort(xhighCeilingModel, Effort.Max)).toBe(Effort.XHigh); + }); + it("returns undefined for reasoning models without controllable efforts (devin-agent shape)", () => { // Repro for https://github.com/can1357/oh-my-pi/issues/3356 — Devin // models report `reasoning: true` but expose no `thinking.efforts` (Cascade @@ -202,13 +221,14 @@ describe("auto thinking classifier helpers", () => { expect(clampAutoThinkingEffort(devinModel, Effort.Low)).toBeUndefined(); expect(clampAutoThinkingEffort(devinModel, Effort.XHigh)).toBeUndefined(); + expect(clampAutoThinkingEffort(devinModel, Effort.Max)).toBeUndefined(); expect(resolveProvisionalAutoLevel(devinModel)).toBeUndefined(); }); - it("accepts max as the top configured thinking alias", () => { - expect(parseEffort("max")).toBe(Effort.XHigh); - expect(parseThinkingLevel("max")).toBeUndefined(); - expect(parseConfiguredThinkingLevel("max")).toBe(ThinkingLevel.XHigh); + it("parses max as a real thinking level", () => { + expect(parseEffort("max")).toBe(Effort.Max); + expect(parseThinkingLevel("max")).toBe(ThinkingLevel.Max); + expect(parseConfiguredThinkingLevel("max")).toBe(ThinkingLevel.Max); }); it("rejects inherited object keys as thinking selectors", () => { diff --git a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts index 40c4d548b..807acd653 100644 --- a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts +++ b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts @@ -52,10 +52,10 @@ describe("parseArgs — --thinking flag", () => { expect(parseArgs(["--thinking=off"]).thinking).toBe(ThinkingLevel.Off); }); - it("accepts auto, concrete efforts, and the max alias", () => { + it("accepts auto and every concrete effort including max", () => { expect(parseArgs(["--thinking", "auto"]).thinking).toBe(AUTO_THINKING); expect(parseArgs(["--thinking", "medium"]).thinking).toBe(Effort.Medium); - expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.XHigh); + expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.Max); }); it("ignores invalid levels and the internal inherit selector", () => { diff --git a/packages/coding-agent/test/cli/completions.test.ts b/packages/coding-agent/test/cli/completions.test.ts index 4da7f19ef..3dd8dd9a6 100644 --- a/packages/coding-agent/test/cli/completions.test.ts +++ b/packages/coding-agent/test/cli/completions.test.ts @@ -211,7 +211,7 @@ describe("omp completions (integration / drift)", () => { } expect(stdout).toContain("{-r,--resume}"); // Real enum option sets flow through unchanged. - expect(stdout).toContain(":value:(off minimal low medium high xhigh auto)"); + expect(stdout).toContain(":value:(off minimal low medium high xhigh max auto)"); expect(stdout).toContain(":value:(always-ask write yolo)"); // Real subcommands present; dynamic callbacks wired. expect(stdout).toContain("_omp_cmd_commit"); diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index e3e36fd96..4e2e5d4fc 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -219,6 +219,25 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [ }), ]; +const mockMaxCapableModels: Model<"anthropic-messages">[] = [ + buildModel({ + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + }, + input: ["text", "image"], + cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 }, + contextWindow: 200000, + maxTokens: 32000, + }), +]; + const openaiGpt55Models: Model[] = [ buildModel({ id: "gpt-5.5", @@ -410,7 +429,15 @@ describe("parseModelPattern", () => { }); test("all valid thinking levels work", () => { - const levels = ["off", Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] as const; + const levels = [ + "off", + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + Effort.Max, + ] as const; for (const level of levels) { const result = parseModelPattern(`sonnet:${level}`, allModels); expect(result.model?.id).toBe("claude-sonnet-4-5"); @@ -418,15 +445,15 @@ describe("parseModelPattern", () => { expect(result.warning).toBeUndefined(); } }); - test("max aliases the highest thinking level after the literal pattern misses", () => { + test("max parses as a real thinking level after the literal pattern misses", () => { const result = parseModelPattern("gpt-5.3-codex:max", allModels); expect(result.model?.id).toBe("gpt-5.3-codex"); - expect(result.thinkingLevel).toBe(Effort.XHigh); + expect(result.thinkingLevel).toBe(Effort.Max); expect(result.explicitThinkingLevel).toBe(true); expect(result.warning).toBeUndefined(); }); - test("literal model ids ending in max win over the thinking alias", () => { + test("literal model ids ending in max win over the thinking suffix", () => { const result = parseModelPattern("nanogpt/coding-router:max", mockMaxSuffixModels); expect(result.model?.id).toBe("coding-router:max"); expect(result.thinkingLevel).toBeUndefined(); @@ -523,13 +550,13 @@ describe("parseModelPattern", () => { expect(result.warning).toBeUndefined(); }); - test("openrouter/:max applies xhigh through the exact-selector path, not an OpenRouter route", () => { - // `max` is a thinking alias, never an OpenRouter route suffix: the request must - // resolve the base model and carry xhigh, not clone a literal `z-ai/glm-4.7:max`. + test("openrouter/:max applies max through the exact-selector path, not an OpenRouter route", () => { + // `max` is a thinking-level suffix, never an OpenRouter route suffix: the request + // must resolve the base model and carry max, not clone a literal `z-ai/glm-4.7:max`. const result = parseModelPattern("openrouter/z-ai/glm-4.7:max", allModels); expect(result.model?.provider).toBe("openrouter"); expect(result.model?.id).toBe("z-ai/glm-4.7"); - expect(result.thinkingLevel).toBe(Effort.XHigh); + expect(result.thinkingLevel).toBe(Effort.Max); expect(result.explicitThinkingLevel).toBe(true); }); }); @@ -618,6 +645,7 @@ describe("resolveModelRoleValue", () => { expect(result.model?.provider).toBe("openai-codex"); expect(result.model?.id).toBe("gpt-5.3-codex"); + // Role-value resolution clamps: gpt-5.3-codex's ladder tops out at xhigh. expect(result.thinkingLevel).toBe(Effort.XHigh); expect(result.explicitThinkingLevel).toBe(true); }); @@ -679,6 +707,15 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); + test("passes max through unclamped when the model ladder includes it", () => { + const result = resolveModelRoleValue("anthropic/claude-opus-4-7:max", mockMaxCapableModels); + + expect(result.model?.provider).toBe("anthropic"); + expect(result.model?.id).toBe("claude-opus-4-7"); + expect(result.thinkingLevel).toBe(Effort.Max); + expect(result.explicitThinkingLevel).toBe(true); + }); + test("preserves an explicit :auto suffix as an explicit thinking selector", () => { const result = resolveModelRoleValue("anthropic/claude-sonnet-4-5:auto", allModels); @@ -1071,7 +1108,7 @@ describe("resolveModelScope", () => { expect(scoped[0].model.id).toBe("gpt-5.5"); }); - test("applies max thinking aliases to glob scopes when no literal max ids match", async () => { + test("applies max thinking selectors to glob scopes when no literal max ids match", async () => { const registry = { getAvailable: () => mockCodexOverlapModels, }; @@ -1079,10 +1116,23 @@ describe("resolveModelScope", () => { const scoped = await resolveModelScope(["openai-codex/*:max"], registry); expect(scoped).toHaveLength(2); + // Scoped levels clamp per model: max on an xhigh-ceiling ladder resolves to xhigh. expect(scoped.map(entry => entry.thinkingLevel)).toEqual([Effort.XHigh, Effort.XHigh]); expect(scoped.every(entry => entry.explicitThinkingLevel)).toBe(true); }); + test("keeps max on glob scopes when the model ladder includes it", async () => { + const registry = { + getAvailable: () => mockMaxCapableModels, + }; + + const scoped = await resolveModelScope(["anthropic/*:max"], registry); + + expect(scoped).toHaveLength(1); + expect(scoped[0].thinkingLevel).toBe(Effort.Max); + expect(scoped[0].explicitThinkingLevel).toBe(true); + }); + test("preserves literal :max in scoped-model globs", async () => { const registry = { getAvailable: () => mockMaxSuffixModels, @@ -1144,18 +1194,25 @@ describe("parseModelString", () => { }); test("extracts max when explicitly enabled for provider id selectors", () => { - const result = parseModelString("deepseek/deepseek-v4-pro:max", { allowMaxAlias: true }); - expect(result).toEqual({ provider: "deepseek", id: "deepseek-v4-pro", thinkingLevel: Effort.XHigh }); + const result = parseModelString("deepseek/deepseek-v4-pro:max", { allowMaxSuffix: true }); + expect(result).toEqual({ provider: "deepseek", id: "deepseek-v4-pro", thinkingLevel: Effort.Max }); }); test("preserves literal max model ids when the caller can prove they exist", () => { const result = parseModelString("nanogpt/coding-router:max", { - allowMaxAlias: true, + allowMaxSuffix: true, isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); expect(result).toEqual({ provider: "nanogpt", id: "coding-router:max" }); }); + test("leaves :max attached to the model id unless the caller opts in via allowMaxSuffix", () => { + // Without allowMaxSuffix, the strict suffix parser must not silently + // reinterpret a literal `:max` id as a thinking suffix. + const result = parseModelString("anthropic/claude-sonnet-4-5:max"); + expect(result).toEqual({ provider: "anthropic", id: "claude-sonnet-4-5:max" }); + }); + test("leaves :auto attached to the model id unless the caller opts in via allowAutoAlias", () => { // Without allowAutoAlias, the strict suffix parser must not silently // reinterpret a literal `:auto` id as an auto-thinking selector. @@ -1242,7 +1299,7 @@ describe("extractExplicitThinkingSelector", () => { const result = extractExplicitThinkingSelector("nanogpt/coding-router:max", undefined, { isLiteralModelId: () => false, }); - expect(result).toBe(Effort.XHigh); + expect(result).toBe(Effort.Max); }); test("treats max on pi role aliases as an explicit selector before expansion", () => { @@ -1251,7 +1308,7 @@ describe("extractExplicitThinkingSelector", () => { const result = extractExplicitThinkingSelector("pi/smol:max", settings, { isLiteralModelId: (provider, id) => provider === "nanogpt" && id === "coding-router:max", }); - expect(result).toBe(Effort.XHigh); + expect(result).toBe(Effort.Max); }); test("does not carry auto from literal role model ids", () => { diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 50db05d61..0b9da2aec 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -171,7 +171,7 @@ describe("ModelSelector role badge thinking display", () => { expect(menuRendered).toContain("Set as SMOL (Quick)"); }); - test("renders xhigh effort for OpenAI GPT-5.5 thinking options", async () => { + test("renders the xhigh-ceiling ladder without a speculative max tier (GPT-5.5)", async () => { installTestTheme(); const model = getBundledModel("openai", "gpt-5.5"); if (!model) throw new Error("Expected bundled model openai/gpt-5.5"); @@ -186,7 +186,25 @@ describe("ModelSelector role badge thinking display", () => { const rendered = normalizeRenderedText(selector.render(220).join("\n")); expect(rendered).toContain("Thinking for: Default (gpt-5.5)"); expect(rendered).toContain("low medium high xhigh"); - expect(rendered).not.toContain("low medium high max"); + // gpt-5.5's wire has no max tier; the selector must not invent one. + expect(rendered).not.toContain("max"); + }); + + test("renders max as a real final tier on max-capable models (GPT-5.6)", async () => { + installTestTheme(); + const model = getBundledModel("openai", "gpt-5.6"); + if (!model) throw new Error("Expected bundled model openai/gpt-5.6"); + + const selector = createSelector(model, Settings.isolated({})); + await Bun.sleep(0); + installTestTheme(); + + selector.handleInput("\n"); + selector.handleInput("\n"); + + const rendered = normalizeRenderedText(selector.render(220).join("\n")); + expect(rendered).toContain("Thinking for: Default (gpt-5.6)"); + expect(rendered).toContain("low medium high xhigh max"); }); test("reloads DEFAULT(auto) from defaultThinkingLevel", async () => { diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index ab3c0396c..e2706d95e 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -277,7 +277,7 @@ describe("createAgentSession deferred model pattern resolution", () => { expect(session.thinkingLevel).toBe("off"); }); - test("normalizes max default thinking level from settings", async () => { + test("clamps a max default thinking level to the model's ladder ceiling", async () => { const settings = Settings.isolated({ defaultThinkingLevel: "max" }); const { session } = await createAgentSession({ @@ -287,6 +287,8 @@ describe("createAgentSession deferred model pattern resolution", () => { expect(session.model?.provider).toBe("runtime-provider"); expect(session.model?.id).toBe("runtime-reasoning-model"); + // The extension model has no explicit ladder; the inferred fallback tops + // out at xhigh, so the real max level clamps down. expect(session.thinkingLevel).toBe(Effort.XHigh); }); diff --git a/packages/terminal-bench/README.md b/packages/terminal-bench/README.md index 1c5786942..4dd12f551 100644 --- a/packages/terminal-bench/README.md +++ b/packages/terminal-bench/README.md @@ -65,7 +65,7 @@ bun src/runner.ts [options] [-- ] | `-n, --concurrency ` | `4` | Concurrent trials | | `-k, --attempts ` | `1` | Attempts per task (pass@k) | | `-i/-x, --include/--exclude ` | — | Task filters (repeatable) | -| `--thinking ` | — | `off…xhigh` | +| `--thinking ` | — | `off…max` | | `--advisor-model

` | — | Second model reviewing the primary; spend summed in | | `--agent ` | `omp` | `oracle`/`nop`/any harbor agent (bypasses omp) | | `--install ` | `local` | `published` = npm `@oh-my-pi/pi-coding-agent` | diff --git a/packages/terminal-bench/src/runner.ts b/packages/terminal-bench/src/runner.ts index 33b24c52e..2aa0ee02c 100755 --- a/packages/terminal-bench/src/runner.ts +++ b/packages/terminal-bench/src/runner.ts @@ -113,7 +113,7 @@ Model / agent: --agent omp (default) | oracle | nop | any harbor agent --install omp source. local = pack /work/pi (default) --version omp version for published install (default: latest) - --thinking off|minimal|low|medium|high|xhigh + --thinking off|minimal|low|medium|high|xhigh|max --advisor-model

Second model reviewing the primary (spend summed in) --advisor-sync Advisor catch-up backlog (default 1 = accurate spend; off = faster) --tarball Reuse a prebuilt omp tarball (implies --no-build) diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index fcb0035da..aebfa8e0a 100755 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -116,7 +116,7 @@ Usage: Options: --model Provider/model ID, e.g. anthropic/claude-sonnet-4-20250514 (default) --provider Override provider (auto-detected from model prefix if omitted) - --thinking Thinking level: off, minimal, low, medium, high, xhigh + --thinking Thinking level: off, minimal, low, medium, high, xhigh, max --runs Runs per task (default: 1) --timeout Timeout per run in ms (default: 120000) --connection-timeout Timeout for first event before fast-retry (default: 30000) diff --git a/python/omp-rpc/src/omp_rpc/protocol.py b/python/omp-rpc/src/omp_rpc/protocol.py index a761e4d34..ccd94cee7 100644 --- a/python/omp-rpc/src/omp_rpc/protocol.py +++ b/python/omp-rpc/src/omp_rpc/protocol.py @@ -11,8 +11,10 @@ JsonValue: TypeAlias = JsonPrimitive | list["JsonValue"] | dict[str, "JsonValue" JsonObject: TypeAlias = dict[str, JsonValue] Attribution: TypeAlias = Literal["user", "agent"] -Effort: TypeAlias = Literal["minimal", "low", "medium", "high", "xhigh"] -ThinkingLevel: TypeAlias = Literal["off", "minimal", "low", "medium", "high", "xhigh"] +Effort: TypeAlias = Literal["minimal", "low", "medium", "high", "xhigh", "max"] +ThinkingLevel: TypeAlias = Literal[ + "off", "minimal", "low", "medium", "high", "xhigh", "max" +] StreamingBehavior: TypeAlias = Literal["steer", "followUp"] SteeringMode: TypeAlias = Literal["all", "one-at-a-time"] InterruptMode: TypeAlias = Literal["immediate", "wait"] @@ -50,7 +52,7 @@ VALUE_EXTENSION_UI_METHODS: Final[frozenset[ValueExtensionUiMethod]] = frozenset {"select", "input", "editor"} ) _EFFORT_VALUES: Final[frozenset[str]] = frozenset( - {"minimal", "low", "medium", "high", "xhigh"} + {"minimal", "low", "medium", "high", "xhigh", "max"} ) _THINKING_LEVEL_VALUES: Final[frozenset[str]] = _EFFORT_VALUES | frozenset({"off"}) _STEERING_MODE_VALUES: Final[frozenset[str]] = frozenset({"all", "one-at-a-time"}) diff --git a/python/omp-rpc/uv.lock b/python/omp-rpc/uv.lock new file mode 100644 index 000000000..3966be0f4 --- /dev/null +++ b/python/omp-rpc/uv.lock @@ -0,0 +1,8 @@ +version = 1 +revision = 3 +requires-python = ">=3.11" + +[[package]] +name = "omp-rpc" +version = "0.1.0" +source = { editable = "." } diff --git a/python/robomp/src/config.py b/python/robomp/src/config.py index 6b645f90c..af6abc285 100644 --- a/python/robomp/src/config.py +++ b/python/robomp/src/config.py @@ -10,7 +10,7 @@ from typing import Literal from pydantic import Field, SecretStr, field_validator, model_validator from pydantic_settings import BaseSettings, SettingsConfigDict -ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh"] +ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh", "max"] class Settings(BaseSettings): diff --git a/python/robomp/src/pragmas.py b/python/robomp/src/pragmas.py index 1f5a9cd92..4f4114aa7 100644 --- a/python/robomp/src/pragmas.py +++ b/python/robomp/src/pragmas.py @@ -33,7 +33,7 @@ Supported keys (today): contains `` (case-insensitive). Falls back to the normal random pool selection if no member matches. - `/thinking ` — override `ROBOMP_THINKING` for this run. Accepts - `off|none|no`, `lo|low`, `med|medium`, `hi|high`, `xhi|xhigh` + `off|none|no`, `lo|low`, `med|medium`, `hi|high`, `xhi|xhigh`, `max` (case-insensitive); anything else is ignored. Parser semantics: @@ -49,7 +49,7 @@ from __future__ import annotations import re from typing import Literal -ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh"] +ThinkingLevel = Literal["off", "low", "medium", "high", "xhigh", "max"] # Key = ascii lowercase / digit / dash / underscore, must start with a letter. # The value (when using `/key=value` form) runs to end-of-token. @@ -164,6 +164,7 @@ _THINKING_ALIASES: dict[str, ThinkingLevel] = { "high": "high", "xhi": "xhigh", "xhigh": "xhigh", + "max": "max", } diff --git a/scripts/edit_benchmark_common.py b/scripts/edit_benchmark_common.py index 2b7c37859..3280fc1a8 100644 --- a/scripts/edit_benchmark_common.py +++ b/scripts/edit_benchmark_common.py @@ -654,7 +654,7 @@ def install_verbose_logging( with _PRINT_LOCK: sys.stderr.write( f"[{model.removeprefix('openrouter/')}] verbose> " - "no thinking level requested; pass --thinking low|medium|high|xhigh if the provider exposes reasoning.\n" + "no thinking level requested; pass --thinking low|medium|high|xhigh|max if the provider exposes reasoning.\n" ) sys.stderr.flush() @@ -914,7 +914,7 @@ def parse_args(description: str) -> argparse.Namespace: ) parser.add_argument( "--thinking", - choices=["off", "minimal", "low", "medium", "high", "xhigh"], + choices=["off", "minimal", "low", "medium", "high", "xhigh", "max"], default="medium", help="Request a specific thinking level for models that support reasoning (default: medium).", )