fix(ai): sanitized Anthropic tool-use arguments with

- Sanitized Anthropic tool-use arguments with toWellFormedDeep on every replay path.
- Updated unsigned-thinking replay tests to assert same-API tool arguments are sanitized.
- Documented the lone-surrogate replay fix in the ai package changelog.
This commit is contained in:
can1357
2026-06-12 15:32:55 +02:00
parent 8874e0f653
commit 4e144be484
3 changed files with 15 additions and 11 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Anthropic-origin tool-call arguments bypassing lone-surrogate sanitization on replay: the model itself can emit unpaired surrogate escapes in its own tool-argument JSON (streamed out fine, then rejected with `400 The request body is not valid JSON` on every subsequent request, bricking the session). `tool_use.input` is now always deep-sanitized with `toWellFormed()`; the pass is identity-preserving, so well-formed arguments stay byte-identical and prompt-cache prefixes are unaffected.
## [15.11.8] - 2026-06-12
### Breaking Changes
+6 -7
View File
@@ -3075,13 +3075,12 @@ export function convertAnthropicMessages(
type: "tool_use",
id: block.id,
name: isOAuthToken ? applyClaudeToolPrefix(block.name) : block.name,
// Anthropic-origin arguments are guaranteed well-formed (they came
// from the API's own JSON); cross-API replays can carry lone
// surrogates that Anthropic's strict UTF-8 validation rejects.
input:
msg.api === "anthropic-messages"
? (block.arguments ?? {})
: toWellFormedDeep(block.arguments ?? {}),
// Always sanitize: the model itself can emit lone-surrogate escapes
// in tool-argument JSON (streamed out fine, rejected with a 400 on
// replay by Anthropic's strict UTF-8 validation). toWellFormedDeep
// is identity-preserving, so well-formed arguments stay
// byte-identical and prompt-cache prefixes are unaffected.
input: toWellFormedDeep(block.arguments ?? {}),
});
}
}
@@ -101,7 +101,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
expect(blocks[1]).toEqual({ type: "text", text: "Sure." });
});
it("sanitizes lone surrogates in cross-API tool arguments only", () => {
it("sanitizes lone surrogates in tool arguments regardless of origin API", () => {
const loneSurrogate = "broken \ud83d end";
const makeToolCallAssistant = (api: AssistantMessage["api"]): AssistantMessage => ({
role: "assistant",
@@ -135,11 +135,12 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
expect(crossToolUse.input.text).toBe("broken \ufffd end");
expect((crossToolUse.input.nested as { parts: string[] }).parts[0]).toBe("broken \ufffd end");
// Same-API replay stays byte-identical (the args came from Anthropic's own
// JSON; rewriting them would destabilize prompt-cache prefixes).
// Same-API replay sanitizes too: the model itself can emit lone-surrogate
// escapes in its own tool-argument JSON (streamed out fine, 400 on replay).
const sameBlocks = assistantWireBlocks([makeUser(), makeToolCallAssistant("anthropic-messages")], makeModel());
const sameToolUse = sameBlocks.find(block => block.type === "tool_use") as WireToolUseBlock;
expect(sameToolUse.input.text).toBe(loneSurrogate);
expect(sameToolUse.input.text).toBe("broken \ufffd end");
expect((sameToolUse.input.nested as { parts: string[] }).parts[0]).toBe("broken \ufffd end");
});
it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => {