fix(ai): overlay compat.reasoningEffortMap onto thinking.effortMap

After PR #2317 moved detected effort maps into model.thinking.effortMap,
the request builders stopped reading the active compat's reasoningEffortMap
at all. That dropped two surfaces:

  1) compat.whenThinking variants whose reasoningEffortMap was applied at
     buildModel time were silently ignored on thinking turns.
  2) Custom raw Model configs that author reasoningEffortMap on compat (no
     buildModel pass) had no path onto the wire.

Introduce a single resolveActiveEffortMap helper in both openai-completions
and openai-responses that overlays the active compat's reasoningEffortMap on
top of model.thinking.effortMap. Default catalog path is unchanged because
the resolved compat.reasoningEffortMap is empty.

Fixes #2315
This commit is contained in:
roboomp
2026-06-11 23:37:49 +00:00
parent efd0812791
commit 2d44d7ff71
3 changed files with 63 additions and 4 deletions
@@ -1372,7 +1372,7 @@ function buildParams(
openRouterParams.reasoning = { enabled: false };
} else if (options?.reasoning) {
openRouterParams.reasoning = {
effort: mapReasoningEffort(options.reasoning, model.thinking?.effortMap),
effort: mapReasoningEffort(options.reasoning, resolveActiveEffortMap(model, compat)),
};
}
} else if (
@@ -1383,7 +1383,7 @@ function buildParams(
compat.supportsReasoningEffort
) {
// OpenAI-style reasoning_effort
params.reasoning_effort = mapReasoningEffort(options.reasoning, model.thinking?.effortMap) as Effort;
params.reasoning_effort = mapReasoningEffort(options.reasoning, resolveActiveEffortMap(model, compat)) as Effort;
} else if (
supportsReasoningParams &&
options?.disableReasoning &&
@@ -1398,7 +1398,7 @@ function buildParams(
if (minEffort === undefined) {
throw new Error(`Model ${model.provider}/${model.id} has no supported reasoning efforts`);
}
params.reasoning_effort = mapReasoningEffort(minEffort, model.thinking?.effortMap) as Effort;
params.reasoning_effort = mapReasoningEffort(minEffort, resolveActiveEffortMap(model, compat)) as Effort;
}
if (compat.disableReasoningOnToolChoice && params.tool_choice !== undefined) {
@@ -1541,6 +1541,23 @@ function mapReasoningEffort(
return reasoningEffortMap?.[effort] ?? effort;
}
/**
* Compose the effective effort-map for the current turn: catalog-baked
* `model.thinking.effortMap` is the default, with the active compat's
* `reasoningEffortMap` overlaid on top so `compat.whenThinking` variants
* (and custom raw `Model` configs) keep authoring their own overrides.
*/
function resolveActiveEffortMap(
model: Model<"openai-completions">,
compat: ResolvedOpenAICompat,
): Partial<Record<NonNullable<OpenAICompletionsOptions["reasoning"]>, string>> | undefined {
const compatMap = compat.reasoningEffortMap;
const thinkingMap = model.thinking?.effortMap;
if (!compatMap || Object.keys(compatMap).length === 0) return thinkingMap;
if (!thinkingMap) return compatMap;
return { ...thinkingMap, ...compatMap };
}
function maybeAddAnthropicCacheControl(compat: ResolvedOpenAICompat, messages: ChatCompletionMessageParam[]): void {
if (compat.cacheControlFormat !== "anthropic") return;
// Anthropic-style caching requires cache_control on a text part. Add a breakpoint
+19 -1
View File
@@ -501,7 +501,10 @@ function buildParams(
options,
messages,
effort =>
mapReasoningEffort(effort as NonNullable<OpenAIResponsesOptions["reasoning"]>, model.thinking?.effortMap),
mapReasoningEffort(
effort as NonNullable<OpenAIResponsesOptions["reasoning"]>,
resolveActiveEffortMap(model),
),
options?.includeEncryptedReasoning ?? true,
options?.omitReasoningEffort ?? false,
);
@@ -520,6 +523,21 @@ function mapReasoningEffort(
return reasoningEffortMap?.[effort] ?? effort;
}
/**
* Compose the effective effort-map: catalog-baked `model.thinking.effortMap`
* is the default, with the resolved compat's `reasoningEffortMap` overlaid on
* top so custom raw `Model` configs keep authoring their own overrides.
*/
function resolveActiveEffortMap(
model: Model<"openai-responses">,
): Partial<Record<NonNullable<OpenAIResponsesOptions["reasoning"]>, string>> | undefined {
const compatMap = model.compat.reasoningEffortMap;
const thinkingMap = model.thinking?.effortMap;
if (!compatMap || Object.keys(compatMap).length === 0) return thinkingMap;
if (!thinkingMap) return compatMap;
return { ...thinkingMap, ...compatMap };
}
function convertConversationMessages(
model: Model<"openai-responses">,
context: Context,
+24
View File
@@ -104,6 +104,30 @@ describe("issue #2315 — MiniMax M2 / GPT-OSS catalog excludes unsupported reas
expect(body.reasoning_effort).toBe("low");
});
it("preserves a custom compat.whenThinking reasoningEffortMap override at request time", async () => {
const base = getBundledModel("fireworks", "minimax-m2.7") as Model<"openai-completions">;
// Custom proxy: minimal stays clamped to low, xhigh is force-mapped to high
// via a whenThinking variant — the swap was getting lost when this PR
// moved effort maps onto `model.thinking.effortMap`.
const model = buildModel({
id: base.id,
name: base.name,
api: "openai-completions",
provider: "fireworks",
baseUrl: base.baseUrl,
reasoning: true,
thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High] },
compat: { whenThinking: { reasoningEffortMap: { high: "max" } } },
input: base.input,
cost: base.cost,
contextWindow: base.contextWindow,
maxTokens: base.maxTokens,
});
const body = await capturePayload(model, { reasoning: Effort.High });
expect(body.reasoning_effort).toBe("max");
});
it("preserves low/medium/high passthrough on fireworks/minimax-m2.7", async () => {
const model = getBundledModel("fireworks", "minimax-m2.7") as Model<"openai-completions">;
const lowBody = await capturePayload(model, { reasoning: Effort.Low });