fix(ai): overlay compat.reasoningEffortMap onto thinking.effortMap
After PR #2317 moved detected effort maps into model.thinking.effortMap, the request builders stopped reading the active compat's reasoningEffortMap at all. That dropped two surfaces: 1) compat.whenThinking variants whose reasoningEffortMap was applied at buildModel time were silently ignored on thinking turns. 2) Custom raw Model configs that author reasoningEffortMap on compat (no buildModel pass) had no path onto the wire. Introduce a single resolveActiveEffortMap helper in both openai-completions and openai-responses that overlays the active compat's reasoningEffortMap on top of model.thinking.effortMap. Default catalog path is unchanged because the resolved compat.reasoningEffortMap is empty. Fixes #2315
This commit is contained in:
@@ -1372,7 +1372,7 @@ function buildParams(
|
||||
openRouterParams.reasoning = { enabled: false };
|
||||
} else if (options?.reasoning) {
|
||||
openRouterParams.reasoning = {
|
||||
effort: mapReasoningEffort(options.reasoning, model.thinking?.effortMap),
|
||||
effort: mapReasoningEffort(options.reasoning, resolveActiveEffortMap(model, compat)),
|
||||
};
|
||||
}
|
||||
} else if (
|
||||
@@ -1383,7 +1383,7 @@ function buildParams(
|
||||
compat.supportsReasoningEffort
|
||||
) {
|
||||
// OpenAI-style reasoning_effort
|
||||
params.reasoning_effort = mapReasoningEffort(options.reasoning, model.thinking?.effortMap) as Effort;
|
||||
params.reasoning_effort = mapReasoningEffort(options.reasoning, resolveActiveEffortMap(model, compat)) as Effort;
|
||||
} else if (
|
||||
supportsReasoningParams &&
|
||||
options?.disableReasoning &&
|
||||
@@ -1398,7 +1398,7 @@ function buildParams(
|
||||
if (minEffort === undefined) {
|
||||
throw new Error(`Model ${model.provider}/${model.id} has no supported reasoning efforts`);
|
||||
}
|
||||
params.reasoning_effort = mapReasoningEffort(minEffort, model.thinking?.effortMap) as Effort;
|
||||
params.reasoning_effort = mapReasoningEffort(minEffort, resolveActiveEffortMap(model, compat)) as Effort;
|
||||
}
|
||||
|
||||
if (compat.disableReasoningOnToolChoice && params.tool_choice !== undefined) {
|
||||
@@ -1541,6 +1541,23 @@ function mapReasoningEffort(
|
||||
return reasoningEffortMap?.[effort] ?? effort;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compose the effective effort-map for the current turn: catalog-baked
|
||||
* `model.thinking.effortMap` is the default, with the active compat's
|
||||
* `reasoningEffortMap` overlaid on top so `compat.whenThinking` variants
|
||||
* (and custom raw `Model` configs) keep authoring their own overrides.
|
||||
*/
|
||||
function resolveActiveEffortMap(
|
||||
model: Model<"openai-completions">,
|
||||
compat: ResolvedOpenAICompat,
|
||||
): Partial<Record<NonNullable<OpenAICompletionsOptions["reasoning"]>, string>> | undefined {
|
||||
const compatMap = compat.reasoningEffortMap;
|
||||
const thinkingMap = model.thinking?.effortMap;
|
||||
if (!compatMap || Object.keys(compatMap).length === 0) return thinkingMap;
|
||||
if (!thinkingMap) return compatMap;
|
||||
return { ...thinkingMap, ...compatMap };
|
||||
}
|
||||
|
||||
function maybeAddAnthropicCacheControl(compat: ResolvedOpenAICompat, messages: ChatCompletionMessageParam[]): void {
|
||||
if (compat.cacheControlFormat !== "anthropic") return;
|
||||
// Anthropic-style caching requires cache_control on a text part. Add a breakpoint
|
||||
|
||||
@@ -501,7 +501,10 @@ function buildParams(
|
||||
options,
|
||||
messages,
|
||||
effort =>
|
||||
mapReasoningEffort(effort as NonNullable<OpenAIResponsesOptions["reasoning"]>, model.thinking?.effortMap),
|
||||
mapReasoningEffort(
|
||||
effort as NonNullable<OpenAIResponsesOptions["reasoning"]>,
|
||||
resolveActiveEffortMap(model),
|
||||
),
|
||||
options?.includeEncryptedReasoning ?? true,
|
||||
options?.omitReasoningEffort ?? false,
|
||||
);
|
||||
@@ -520,6 +523,21 @@ function mapReasoningEffort(
|
||||
return reasoningEffortMap?.[effort] ?? effort;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compose the effective effort-map: catalog-baked `model.thinking.effortMap`
|
||||
* is the default, with the resolved compat's `reasoningEffortMap` overlaid on
|
||||
* top so custom raw `Model` configs keep authoring their own overrides.
|
||||
*/
|
||||
function resolveActiveEffortMap(
|
||||
model: Model<"openai-responses">,
|
||||
): Partial<Record<NonNullable<OpenAIResponsesOptions["reasoning"]>, string>> | undefined {
|
||||
const compatMap = model.compat.reasoningEffortMap;
|
||||
const thinkingMap = model.thinking?.effortMap;
|
||||
if (!compatMap || Object.keys(compatMap).length === 0) return thinkingMap;
|
||||
if (!thinkingMap) return compatMap;
|
||||
return { ...thinkingMap, ...compatMap };
|
||||
}
|
||||
|
||||
function convertConversationMessages(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
|
||||
@@ -104,6 +104,30 @@ describe("issue #2315 — MiniMax M2 / GPT-OSS catalog excludes unsupported reas
|
||||
expect(body.reasoning_effort).toBe("low");
|
||||
});
|
||||
|
||||
it("preserves a custom compat.whenThinking reasoningEffortMap override at request time", async () => {
|
||||
const base = getBundledModel("fireworks", "minimax-m2.7") as Model<"openai-completions">;
|
||||
// Custom proxy: minimal stays clamped to low, xhigh is force-mapped to high
|
||||
// via a whenThinking variant — the swap was getting lost when this PR
|
||||
// moved effort maps onto `model.thinking.effortMap`.
|
||||
const model = buildModel({
|
||||
id: base.id,
|
||||
name: base.name,
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
baseUrl: base.baseUrl,
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High] },
|
||||
compat: { whenThinking: { reasoningEffortMap: { high: "max" } } },
|
||||
input: base.input,
|
||||
cost: base.cost,
|
||||
contextWindow: base.contextWindow,
|
||||
maxTokens: base.maxTokens,
|
||||
});
|
||||
|
||||
const body = await capturePayload(model, { reasoning: Effort.High });
|
||||
expect(body.reasoning_effort).toBe("max");
|
||||
});
|
||||
|
||||
it("preserves low/medium/high passthrough on fireworks/minimax-m2.7", async () => {
|
||||
const model = getBundledModel("fireworks", "minimax-m2.7") as Model<"openai-completions">;
|
||||
const lowBody = await capturePayload(model, { reasoning: Effort.Low });
|
||||
|
||||
Reference in New Issue
Block a user