feat(ai): ensure Anthropic requests use correct thinking model ID
- Previously, `anthropic-messages` requests using `resolveWireModelId` would always derive the non-thinking variant for `requestModelId`, even when reasoning was explicitly enabled. - This change ensures the `reasoning` state is correctly passed to `resolveWireModelId`, allowing the API request to include the appropriate `X-thinking` model variant when thinking is active, standardizing behavior across providers.
This commit is contained in:
@@ -10,8 +10,8 @@
|
||||
### Changed
|
||||
|
||||
- Changed `google-gemini-cli` request mapping to route per-request wire ids via `resolveWireModelId`: the session effort picks the backing variant id (collapsed `gemini-3.5-flash` at high → `gemini-3.5-flash-low`; claude pairs route off → bare id, efforts → `-thinking`) while `AssistantMessage.model` and usage attribution stay on the logical id. A thinking budget clamped to zero now falls through to the thinking-off path (off routing plus suppression) instead of only disabling thinking
|
||||
|
||||
- Changed `openai-completions` and `anthropic-messages` to serialize per-request wire ids via `resolveWireModelId`, so collapsed `X`/`X-thinking` pairs on aggregators and custom providers switch to the thinking SKU when reasoning is enabled (previously only `google-gemini-cli` routed effort-tier variants)
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `google-gemini-cli` ignoring `Model.requestModelId` when serializing the request model id
|
||||
|
||||
@@ -737,6 +737,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (!reasoning || !model.reasoning) {
|
||||
return castApi<"anthropic-messages">({
|
||||
...base,
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
thinkingEnabled: false,
|
||||
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
||||
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
||||
@@ -748,6 +749,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (thinkingBudget <= 0) {
|
||||
return castApi<"anthropic-messages">({
|
||||
...base,
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
thinkingEnabled: false,
|
||||
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
||||
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
||||
@@ -761,24 +763,24 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
const effort = mapEffortToAnthropicAdaptiveEffort(model, reasoning);
|
||||
return castApi<"anthropic-messages">({
|
||||
...base,
|
||||
requestModelId: resolveWireModelId(model, reasoning),
|
||||
thinkingEnabled: true,
|
||||
effort,
|
||||
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
||||
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
||||
serviceTier: options?.serviceTier,
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
});
|
||||
}
|
||||
|
||||
if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
|
||||
return castApi<"anthropic-messages">({
|
||||
...base,
|
||||
requestModelId: resolveWireModelId(model, reasoning),
|
||||
thinkingEnabled: true,
|
||||
thinkingBudgetTokens: thinkingBudget,
|
||||
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
||||
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
||||
serviceTier: options?.serviceTier,
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
});
|
||||
}
|
||||
|
||||
@@ -792,9 +794,9 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
|
||||
// If thinking budget is too low, disable thinking
|
||||
if (thinkingBudget <= 0) {
|
||||
requestModelId: resolveWireModelId(model, reasoning),
|
||||
return castApi<"anthropic-messages">({
|
||||
...base,
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
thinkingEnabled: false,
|
||||
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
||||
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
||||
@@ -825,7 +827,6 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
// Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
|
||||
if (model.thinking?.mode === "anthropic-adaptive") {
|
||||
return castApi<"bedrock-converse-stream">(bedrockBase);
|
||||
requestModelId: resolveWireModelId(model, undefined),
|
||||
}
|
||||
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
|
||||
if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
|
||||
@@ -835,7 +836,6 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
const desiredMaxTokens = Math.min(model.maxTokens, budgetInfo.budget + MIN_OUTPUT_TOKENS);
|
||||
if (desiredMaxTokens > maxTokens) {
|
||||
maxTokens = desiredMaxTokens;
|
||||
requestModelId: resolveWireModelId(model, reasoning),
|
||||
}
|
||||
}
|
||||
if (maxTokens <= budgetInfo.budget) {
|
||||
|
||||
@@ -48,6 +48,12 @@ SHAPES = {
|
||||
"lineRepeat": 2, "frameSize": SIZE, "frameTokenEstimate": 3300},
|
||||
"8x8r-sent": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "sent",
|
||||
"lineRepeat": 2, "frameSize": SIZE, "frameTokenEstimate": 1100},
|
||||
"8x8u-bw": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "bw",
|
||||
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
|
||||
"8x8u-sent": {"font": "8x8", "cellWidth": 8, "cellHeight": 8, "variant": "sent",
|
||||
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
|
||||
"6x6u-sent": {"font": "8x8", "cellWidth": 6, "cellHeight": 6, "variant": "sent",
|
||||
"lineRepeat": 1, "frameSize": SIZE, "frameTokenEstimate": 3300},
|
||||
}
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user