feat(ai): added JSON healing for literal escapes and trailing junk in LLM output

- Added automatic cleaning of literal escape sequences (`\n`, `\t`, `\r`) in JSON parsing to handle LLM encoding confusion.
- Added support for healing JSON with trailing junk after balanced containers (e.g., `]\n</invoke>`).
- Implemented `cleanLiteralEscapes()` function to distinguish between literal backslash sequences outside JSON strings and actual escape sequences within strings.
- Added 2 test cases validating escape sequence cleaning and trailing junk recovery in tool argument coercion.
This commit is contained in:
can1357
2026-04-08 09:46:40 +02:00
parent 6e19642315
commit bf97331311
3 changed files with 104 additions and 2 deletions
+3
View File
@@ -1,12 +1,15 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Removed `coerceNullStrings` function and its automatic null-string coercion behavior from JSON parsing
### Added
- Added automatic cleaning of literal escape sequences (`\n`, `\t`, `\r`) in JSON parsing to handle LLM encoding confusion
- Added support for healing JSON with trailing junk after balanced containers (e.g., `]\n</invoke>`)
- Added `CODEX_STARTUP_EVENT_CHANNEL` constant and `CodexStartupEvent` type for monitoring Codex provider initialization status
- Added automatic healing of malformed JSON with single-character bracket errors at the end of strings, improving LLM tool argument parsing robustness
+55 -2
View File
@@ -128,16 +128,69 @@ function tryParseLeadingJsonContainer(value: string): unknown | undefined {
depth -= 1;
if (depth !== 0) continue;
const prefix = value.slice(0, index + 1);
try {
return JSON.parse(value.slice(0, index + 1)) as unknown;
return JSON.parse(prefix) as unknown;
} catch {
return undefined;
// LLMs sometimes emit literal `\n` or `\t` between JSON tokens
// (e.g. `[{...}\n]`). Convert these to real whitespace and retry.
const cleaned = cleanLiteralEscapes(prefix);
if (cleaned !== prefix) {
try {
return JSON.parse(cleaned) as unknown;
} catch {}
}
// Also try single-char healing on the extracted prefix.
return tryHealMalformedJson(prefix);
}
}
return undefined;
}
/**
* Replace literal `\n`, `\t`, `\r` sequences that appear OUTSIDE of JSON
* strings with actual whitespace. LLMs sometimes produce these when they
* confuse the tool-call encoding with the content encoding.
*/
function cleanLiteralEscapes(value: string): string {
let result = "";
let inString = false;
let i = 0;
while (i < value.length) {
const ch = value[i];
if (inString) {
if (ch === "\\" && i + 1 < value.length) {
result += ch + value[i + 1];
i += 2;
continue;
}
if (ch === '"') inString = false;
result += ch;
i += 1;
continue;
}
if (ch === '"') {
inString = true;
result += ch;
i += 1;
continue;
}
// Outside a string: replace literal \n, \t, \r with whitespace
if (ch === "\\" && i + 1 < value.length) {
const next = value[i + 1];
if (next === "n" || next === "t" || next === "r") {
result += " ";
i += 2;
continue;
}
}
result += ch;
i += 1;
}
return result;
}
/** Maximum single-character edits to attempt when healing malformed JSON. */
const MAX_HEAL_DISTANCE = 3;
const BRACKET_CHARS = ["[", "]", "{", "}"] as const;
@@ -541,6 +541,52 @@ describe("Tool argument coercion", () => {
expect(result.edits).toEqual([{ target: "fn_bar#1234", content: "return 1}" }]);
});
it("heals stringified array with literal backslash-n between tokens", () => {
const tool: Tool = {
name: "heal-esc-1",
description: "",
parameters: Type.Object({
edits: Type.Array(Type.Object({ target: Type.String(), content: Type.String() })),
}),
};
// LLM emits literal \n between the closing } and ]
const toolCall: ToolCall = {
type: "toolCall",
id: "call-heal-esc-1",
name: "heal-esc-1",
arguments: {
edits: '[{"target": "fn_foo#ABCD@inner", "content": "return 1;\\n"}\\n]',
},
};
const result = validateToolArguments(tool, toolCall);
expect(result.edits).toEqual([{ target: "fn_foo#ABCD@inner", content: "return 1;\n" }]);
});
it("heals stringified array with trailing junk after balanced container", () => {
const tool: Tool = {
name: "heal-trail-1",
description: "",
parameters: Type.Object({
edits: Type.Array(Type.Object({ target: Type.String(), op: Type.String() })),
}),
};
// LLM appends \n</invoke> after the valid JSON
const toolCall: ToolCall = {
type: "toolCall",
id: "call-heal-trail-1",
name: "heal-trail-1",
arguments: {
edits: '[{"target": "fn_foo", "op": "replace"}]\n</invoke>',
},
};
const result = validateToolArguments(tool, toolCall);
expect(result.edits).toEqual([{ target: "fn_foo", op: "replace" }]);
});
it("does not heal deeply broken JSON strings", () => {
const tool: Tool = {
name: "heal-3",