feat(ai): ensure Anthropic requests use correct thinking model ID

- Previously, `anthropic-messages` requests using `resolveWireModelId` would always derive the non-thinking variant for `requestModelId`, even when reasoning was explicitly enabled.
- This change ensures the `reasoning` state is correctly passed to `resolveWireModelId`, allowing the API request to include the appropriate `X-thinking` model variant when thinking is active, standardizing behavior across providers.
This commit is contained in:
can1357
2026-06-12 08:24:23 +02:00
parent 870d2b8981
commit 1308f654a0
3 changed files with 12 additions and 6 deletions
+1 -1
View File
@@ -10,8 +10,8 @@
### Changed
- Changed `google-gemini-cli` request mapping to route per-request wire ids via `resolveWireModelId`: the session effort picks the backing variant id (collapsed `gemini-3.5-flash` at high → `gemini-3.5-flash-low`; claude pairs route off → bare id, efforts → `-thinking`) while `AssistantMessage.model` and usage attribution stay on the logical id. A thinking budget clamped to zero now falls through to the thinking-off path (off routing plus suppression) instead of only disabling thinking
- Changed `openai-completions` and `anthropic-messages` to serialize per-request wire ids via `resolveWireModelId`, so collapsed `X`/`X-thinking` pairs on aggregators and custom providers switch to the thinking SKU when reasoning is enabled (previously only `google-gemini-cli` routed effort-tier variants)
### Fixed
- Fixed `google-gemini-cli` ignoring `Model.requestModelId` when serializing the request model id
+5 -5
View File
@@ -737,6 +737,7 @@ function mapOptionsForApi<TApi extends Api>(
if (!reasoning || !model.reasoning) {
return castApi<"anthropic-messages">({
...base,
requestModelId: resolveWireModelId(model, undefined),
thinkingEnabled: false,
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
@@ -748,6 +749,7 @@ function mapOptionsForApi<TApi extends Api>(
if (thinkingBudget <= 0) {
return castApi<"anthropic-messages">({
...base,
requestModelId: resolveWireModelId(model, undefined),
thinkingEnabled: false,
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
@@ -761,24 +763,24 @@ function mapOptionsForApi<TApi extends Api>(
const effort = mapEffortToAnthropicAdaptiveEffort(model, reasoning);
return castApi<"anthropic-messages">({
...base,
requestModelId: resolveWireModelId(model, reasoning),
thinkingEnabled: true,
effort,
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
serviceTier: options?.serviceTier,
requestModelId: resolveWireModelId(model, undefined),
});
}
if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
return castApi<"anthropic-messages">({
...base,
requestModelId: resolveWireModelId(model, reasoning),
thinkingEnabled: true,
thinkingBudgetTokens: thinkingBudget,
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
serviceTier: options?.serviceTier,
requestModelId: resolveWireModelId(model, undefined),
});
}
@@ -792,9 +794,9 @@ function mapOptionsForApi<TApi extends Api>(
// If thinking budget is too low, disable thinking
if (thinkingBudget <= 0) {
requestModelId: resolveWireModelId(model, reasoning),
return castApi<"anthropic-messages">({
...base,
requestModelId: resolveWireModelId(model, undefined),
thinkingEnabled: false,
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
@@ -825,7 +827,6 @@ function mapOptionsForApi<TApi extends Api>(
// Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
if (model.thinking?.mode === "anthropic-adaptive") {
return castApi<"bedrock-converse-stream">(bedrockBase);
requestModelId: resolveWireModelId(model, undefined),
}
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
@@ -835,7 +836,6 @@ function mapOptionsForApi<TApi extends Api>(
const desiredMaxTokens = Math.min(model.maxTokens, budgetInfo.budget + MIN_OUTPUT_TOKENS);
if (desiredMaxTokens > maxTokens) {
maxTokens = desiredMaxTokens;
requestModelId: resolveWireModelId(model, reasoning),
}
}
if (maxTokens <= budgetInfo.budget) {
@@ -48,6 +48,12 @@ SHAPES = {
"lineRepeat": 2, "frameSize": SIZE, "frameTokenEstimate": 3300},
"8x8r-sent": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "sent",
"lineRepeat": 2, "frameSize": SIZE, "frameTokenEstimate": 1100},
"8x8u-bw": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "bw",
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
"8x8u-sent": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "sent",
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
"6x6u-sent": {"font": "8x8", "cellWidth": 6, "cellHeight": 6, "variant": "sent",
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
}