- Migrated sanitizeText from pi-natives to pi-utils as a pure-JS implementation, removing the native dependency across all call sites.
39 lines
1.3 KiB
TypeScript
39 lines
1.3 KiB
TypeScript
/**
|
|
* Strip ANSI escape sequences, remove control characters / lone surrogates,
|
|
* and normalize line endings.
|
|
*
|
|
* Bun-native implementation of the former native `sanitizeText` (see
|
|
* `crates/pi-natives/src/text.rs::sanitize_text`). JavaScript strings are
|
|
* already UTF-16 code-unit arrays. `toWellFormed()` handles the uncommon
|
|
* malformed path; when it changes the input, replacement characters are
|
|
* dropped and the normalized result goes through the well-formed sanitizer.
|
|
*
|
|
* Fast path: well-formed input with no controls or ANSI returns the original
|
|
* string after the control probe.
|
|
*/
|
|
|
|
const ESC_CHAR = "\x1b";
|
|
|
|
// Well-formed strings only need control/ANSI detection: C0 (excl. \t \n),
|
|
// CR, DEL, and C1. ESC (0x1B) is in \x0B-\x1F.
|
|
const CONTROL_RE = /[\x00-\x08\x0B-\x1F\x7F-\x9F]/g;
|
|
|
|
const REPLACEMENT_CHAR = "\ufffd";
|
|
|
|
export function sanitizeText(text: string): string {
|
|
const wellFormed = text.toWellFormed();
|
|
if (wellFormed !== text) {
|
|
return sanitizeWellFormedText(wellFormed.replaceAll(REPLACEMENT_CHAR, ""));
|
|
}
|
|
return sanitizeWellFormedText(text);
|
|
}
|
|
|
|
function sanitizeWellFormedText(text: string): string {
|
|
CONTROL_RE.lastIndex = 0;
|
|
if (CONTROL_RE.exec(text) === null) return text;
|
|
|
|
const stripped = text.indexOf(ESC_CHAR) === -1 ? text : Bun.stripANSI(text);
|
|
CONTROL_RE.lastIndex = 0;
|
|
return stripped.replace(CONTROL_RE, "");
|
|
}
|