feat(ai): added JSON healing for literal escapes and trailing junk in LLM output
- Added automatic cleaning of literal escape sequences (`\n`, `\t`, `\r`) in JSON parsing to handle LLM encoding confusion. - Added support for healing JSON with trailing junk after balanced containers (e.g., `]\n</invoke>`). - Implemented `cleanLiteralEscapes()` function to distinguish between literal backslash sequences outside JSON strings and actual escape sequences within strings. - Added 2 test cases validating escape sequence cleaning and trailing junk recovery in tool argument coercion.
This commit is contained in:
@@ -1,12 +1,15 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Removed `coerceNullStrings` function and its automatic null-string coercion behavior from JSON parsing
|
||||
|
||||
### Added
|
||||
|
||||
- Added automatic cleaning of literal escape sequences (`\n`, `\t`, `\r`) in JSON parsing to handle LLM encoding confusion
|
||||
- Added support for healing JSON with trailing junk after balanced containers (e.g., `]\n</invoke>`)
|
||||
- Added `CODEX_STARTUP_EVENT_CHANNEL` constant and `CodexStartupEvent` type for monitoring Codex provider initialization status
|
||||
- Added automatic healing of malformed JSON with single-character bracket errors at the end of strings, improving LLM tool argument parsing robustness
|
||||
|
||||
|
||||
@@ -128,16 +128,69 @@ function tryParseLeadingJsonContainer(value: string): unknown | undefined {
|
||||
depth -= 1;
|
||||
if (depth !== 0) continue;
|
||||
|
||||
const prefix = value.slice(0, index + 1);
|
||||
try {
|
||||
return JSON.parse(value.slice(0, index + 1)) as unknown;
|
||||
return JSON.parse(prefix) as unknown;
|
||||
} catch {
|
||||
return undefined;
|
||||
// LLMs sometimes emit literal `\n` or `\t` between JSON tokens
|
||||
// (e.g. `[{...}\n]`). Convert these to real whitespace and retry.
|
||||
const cleaned = cleanLiteralEscapes(prefix);
|
||||
if (cleaned !== prefix) {
|
||||
try {
|
||||
return JSON.parse(cleaned) as unknown;
|
||||
} catch {}
|
||||
}
|
||||
// Also try single-char healing on the extracted prefix.
|
||||
return tryHealMalformedJson(prefix);
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Replace literal `\n`, `\t`, `\r` sequences that appear OUTSIDE of JSON
|
||||
* strings with actual whitespace. LLMs sometimes produce these when they
|
||||
* confuse the tool-call encoding with the content encoding.
|
||||
*/
|
||||
function cleanLiteralEscapes(value: string): string {
|
||||
let result = "";
|
||||
let inString = false;
|
||||
let i = 0;
|
||||
while (i < value.length) {
|
||||
const ch = value[i];
|
||||
if (inString) {
|
||||
if (ch === "\\" && i + 1 < value.length) {
|
||||
result += ch + value[i + 1];
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') inString = false;
|
||||
result += ch;
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') {
|
||||
inString = true;
|
||||
result += ch;
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
// Outside a string: replace literal \n, \t, \r with whitespace
|
||||
if (ch === "\\" && i + 1 < value.length) {
|
||||
const next = value[i + 1];
|
||||
if (next === "n" || next === "t" || next === "r") {
|
||||
result += " ";
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
result += ch;
|
||||
i += 1;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/** Maximum single-character edits to attempt when healing malformed JSON. */
|
||||
const MAX_HEAL_DISTANCE = 3;
|
||||
const BRACKET_CHARS = ["[", "]", "{", "}"] as const;
|
||||
|
||||
@@ -541,6 +541,52 @@ describe("Tool argument coercion", () => {
|
||||
expect(result.edits).toEqual([{ target: "fn_bar#1234", content: "return 1}" }]);
|
||||
});
|
||||
|
||||
it("heals stringified array with literal backslash-n between tokens", () => {
|
||||
const tool: Tool = {
|
||||
name: "heal-esc-1",
|
||||
description: "",
|
||||
parameters: Type.Object({
|
||||
edits: Type.Array(Type.Object({ target: Type.String(), content: Type.String() })),
|
||||
}),
|
||||
};
|
||||
|
||||
// LLM emits literal \n between the closing } and ]
|
||||
const toolCall: ToolCall = {
|
||||
type: "toolCall",
|
||||
id: "call-heal-esc-1",
|
||||
name: "heal-esc-1",
|
||||
arguments: {
|
||||
edits: '[{"target": "fn_foo#ABCD@inner", "content": "return 1;\\n"}\\n]',
|
||||
},
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, toolCall);
|
||||
expect(result.edits).toEqual([{ target: "fn_foo#ABCD@inner", content: "return 1;\n" }]);
|
||||
});
|
||||
|
||||
it("heals stringified array with trailing junk after balanced container", () => {
|
||||
const tool: Tool = {
|
||||
name: "heal-trail-1",
|
||||
description: "",
|
||||
parameters: Type.Object({
|
||||
edits: Type.Array(Type.Object({ target: Type.String(), op: Type.String() })),
|
||||
}),
|
||||
};
|
||||
|
||||
// LLM appends \n</invoke> after the valid JSON
|
||||
const toolCall: ToolCall = {
|
||||
type: "toolCall",
|
||||
id: "call-heal-trail-1",
|
||||
name: "heal-trail-1",
|
||||
arguments: {
|
||||
edits: '[{"target": "fn_foo", "op": "replace"}]\n</invoke>',
|
||||
},
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, toolCall);
|
||||
expect(result.edits).toEqual([{ target: "fn_foo", op: "replace" }]);
|
||||
});
|
||||
|
||||
it("does not heal deeply broken JSON strings", () => {
|
||||
const tool: Tool = {
|
||||
name: "heal-3",
|
||||
|
||||
Reference in New Issue
Block a user