diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 5b1ebd1ff..2b365a4e8 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -20,6 +20,7 @@ - Scoped Codex reactive backoff per meter: a `usage_limit_reached` from a Spark request no longer persists a block that ordinary chat requests honour, and the reverse. Blocks written before scoping used a shared scope meaning "block everything", so requests still honour it and reconciliation still heals it - Implemented `scopeLimits` for the Codex ranking strategy so a request gates only on the windows it actually consumes: `-spark` models spend the Spark meter and every other model spends the 5h/weekly chat windows, instead of OR-ing every window and meter into one provider-wide block - Fixed native Anthropic adaptive-only models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) keeping thinking ON when reasoning was meant to be off. `mapOptionsForApi` never consulted `disableReasoning` on the Anthropic branch, so a caller-side disable left adaptive thinking at full effort; and `disableThinkingIfToolChoiceForced` deleted `output_config.effort` alongside `thinking`, which for adaptive-only models silently re-enabled adaptive thinking (a bare omission defaults to adaptive-ON). Both paths now omit `thinking` and pin the lowest adaptive effort, so `disableReasoning` and forced `tool_choice` turns (e.g. the delivery reviewer's `report_delivery`) actually suppress reasoning instead of returning a thinking block with `end_turn` ([#6589](https://github.com/can1357/oh-my-pi/issues/6589)). +- Fixed Bedrock Converse dropping captured Claude thinking signatures when replaying application-inference-profile ARN models, restoring adaptive-thinking multi-turn conversations ([#6610](https://github.com/can1357/oh-my-pi/issues/6610)). ## [17.1.3] - 2026-07-24 diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 036575b05..61800b09b 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -830,10 +830,9 @@ function convertMessages( case "thinking": // Skip empty thinking blocks if (c.thinking.trim().length === 0) continue; - // Thinking blocks require a valid signature when sent as reasoningContent. - // If the signature is missing (e.g., from an aborted stream), or the model - // doesn't support signatures, convert to plain text instead. - if (supportsThinkingSignature(model) && c.thinkingSignature) { + // A captured signature is authoritative even when the model id is an opaque ARN. + // Without one, known non-Claude families use unsigned reasoning; known Claude ids demote to text. + if (c.thinkingSignature) { contentBlocks.push({ reasoningContent: { reasoningText: { text: c.thinking.toWellFormed(), signature: c.thinkingSignature }, diff --git a/packages/ai/test/bedrock-inference-profile.test.ts b/packages/ai/test/bedrock-inference-profile.test.ts index 08314bb84..495ad7a44 100644 --- a/packages/ai/test/bedrock-inference-profile.test.ts +++ b/packages/ai/test/bedrock-inference-profile.test.ts @@ -2,6 +2,7 @@ import { describe, expect, test } from "bun:test"; import { streamBedrock } from "@oh-my-pi/pi-ai/providers/amazon-bedrock"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { withEnv } from "./helpers"; const profileArn = "arn:aws:bedrock:us-east-2:1234567890:application-inference-profile/company-opus-48"; @@ -16,6 +17,11 @@ const profileModel: Model<"bedrock-converse-stream"> = buildModel({ cost: { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 }, contextWindow: 1000000, maxTokens: 128000, + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.Max], + supportsDisplay: true, + }, }); function userContext(): Context { @@ -46,6 +52,72 @@ describe("Bedrock inference profile ARNs", () => { `https://bedrock-runtime.us-east-2.amazonaws.com/model/${encodeURIComponent(profileArn)}/converse-stream`, ]); }); + + test("replays captured thinking signatures for ARN profiles", async () => { + const context: Context = { + messages: [ + { role: "user", content: "Plan the change", timestamp: 0 }, + { + role: "assistant", + content: [ + { type: "thinking", thinking: "Inspect the implementation", thinkingSignature: "signed-reasoning" }, + { type: "text", text: "I found the relevant code." }, + ], + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + model: profileArn, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 1, + }, + { role: "user", content: "Continue", timestamp: 2 }, + ], + }; + const controller = new AbortController(); + controller.abort(); + const { promise, resolve } = Promise.withResolvers(); + + void streamBedrock(profileModel, context, { + signal: controller.signal, + reasoning: Effort.High, + maxTokens: 16, + onPayload: payload => { + resolve(payload); + }, + }); + + expect(await promise).toMatchObject({ + additionalModelRequestFields: { + thinking: { type: "adaptive", display: "summarized" }, + output_config: { effort: "high" }, + }, + messages: [ + { role: "user", content: [{ text: "Plan the change" }] }, + { + role: "assistant", + content: [ + { + reasoningContent: { + reasoningText: { + text: "Inspect the implementation", + signature: "signed-reasoning", + }, + }, + }, + { text: "I found the relevant code." }, + ], + }, + { role: "user", content: [{ text: "Continue" }] }, + ], + }); + }); }); function bedrockModel(id: string): Model<"bedrock-converse-stream"> {