fix(ai): sanitized Anthropic tool-use arguments with
- Sanitized Anthropic tool-use arguments with toWellFormedDeep on every replay path. - Updated unsigned-thinking replay tests to assert same-API tool arguments are sanitized. - Documented the lone-surrogate replay fix in the ai package changelog.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Anthropic-origin tool-call arguments bypassing lone-surrogate sanitization on replay: the model itself can emit unpaired surrogate escapes in its own tool-argument JSON (streamed out fine, then rejected with `400 The request body is not valid JSON` on every subsequent request, bricking the session). `tool_use.input` is now always deep-sanitized with `toWellFormed()`; the pass is identity-preserving, so well-formed arguments stay byte-identical and prompt-cache prefixes are unaffected.
|
||||
|
||||
## [15.11.8] - 2026-06-12
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
@@ -3075,13 +3075,12 @@ export function convertAnthropicMessages(
|
||||
type: "tool_use",
|
||||
id: block.id,
|
||||
name: isOAuthToken ? applyClaudeToolPrefix(block.name) : block.name,
|
||||
// Anthropic-origin arguments are guaranteed well-formed (they came
|
||||
// from the API's own JSON); cross-API replays can carry lone
|
||||
// surrogates that Anthropic's strict UTF-8 validation rejects.
|
||||
input:
|
||||
msg.api === "anthropic-messages"
|
||||
? (block.arguments ?? {})
|
||||
: toWellFormedDeep(block.arguments ?? {}),
|
||||
// Always sanitize: the model itself can emit lone-surrogate escapes
|
||||
// in tool-argument JSON (streamed out fine, rejected with a 400 on
|
||||
// replay by Anthropic's strict UTF-8 validation). toWellFormedDeep
|
||||
// is identity-preserving, so well-formed arguments stay
|
||||
// byte-identical and prompt-cache prefixes are unaffected.
|
||||
input: toWellFormedDeep(block.arguments ?? {}),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -101,7 +101,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
expect(blocks[1]).toEqual({ type: "text", text: "Sure." });
|
||||
});
|
||||
|
||||
it("sanitizes lone surrogates in cross-API tool arguments only", () => {
|
||||
it("sanitizes lone surrogates in tool arguments regardless of origin API", () => {
|
||||
const loneSurrogate = "broken \ud83d end";
|
||||
const makeToolCallAssistant = (api: AssistantMessage["api"]): AssistantMessage => ({
|
||||
role: "assistant",
|
||||
@@ -135,11 +135,12 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
expect(crossToolUse.input.text).toBe("broken \ufffd end");
|
||||
expect((crossToolUse.input.nested as { parts: string[] }).parts[0]).toBe("broken \ufffd end");
|
||||
|
||||
// Same-API replay stays byte-identical (the args came from Anthropic's own
|
||||
// JSON; rewriting them would destabilize prompt-cache prefixes).
|
||||
// Same-API replay sanitizes too: the model itself can emit lone-surrogate
|
||||
// escapes in its own tool-argument JSON (streamed out fine, 400 on replay).
|
||||
const sameBlocks = assistantWireBlocks([makeUser(), makeToolCallAssistant("anthropic-messages")], makeModel());
|
||||
const sameToolUse = sameBlocks.find(block => block.type === "tool_use") as WireToolUseBlock;
|
||||
expect(sameToolUse.input.text).toBe(loneSurrogate);
|
||||
expect(sameToolUse.input.text).toBe("broken \ufffd end");
|
||||
expect((sameToolUse.input.nested as { parts: string[] }).parts[0]).toBe("broken \ufffd end");
|
||||
});
|
||||
|
||||
it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => {
|
||||
|
||||
Reference in New Issue
Block a user