feat(text): refactored text processing with UTF-16 optimization and ANSI state bitflags

- Added `visibleWidth()` function to measure visible text width while excluding ANSI codes.
- Optimized text processing from UTF-8 to UTF-16 implementation with direct JsString handling for improved performance.
- Refactored ANSI state tracking to use bitflags-based SgrState struct instead of individual boolean fields for more efficient storage.
- Introduced ColorCode enum to represent color codes (Basic, Indexed, Rgb) for better type safety.
- Added comprehensive test coverage for UTF-16 text processing, ANSI detection, and width calculations.
- Updated napi feature from 'napi8' to 'napi10' and added unicode-segmentation dependency.
This commit is contained in:
can1357
2026-02-01 09:34:54 +01:00
parent b91ab4bad0
commit c8f106c58c
11 changed files with 982 additions and 453 deletions
+12 -36
View File
@@ -33,7 +33,7 @@ const widthCache = new Map<string, number>();
/**
* Calculate the visible width of a string in terminal columns.
*/
export function visibleWidth(str: string): number {
export function visibleWidthRaw(str: string): number {
if (str.length === 0) {
return 0;
}
@@ -52,6 +52,16 @@ export function visibleWidth(str: string): number {
if (isPureAscii) {
return str.length + tabLength;
}
return Bun.stringWidth(str) + tabLength;
}
/**
* Calculate the visible width of a string in terminal columns.
*/
export function visibleWidth(str: string): number {
if (str.length === 0) {
return 0;
}
// Check cache
const cached = widthCache.get(str);
@@ -59,8 +69,7 @@ export function visibleWidth(str: string): number {
return cached;
}
// Cache result
const width = Bun.stringWidth(str) + tabLength;
const width = visibleWidthRaw(str);
if (widthCache.size >= WIDTH_CACHE_SIZE) {
const firstKey = widthCache.keys().next().value;
if (firstKey !== undefined) {
@@ -72,39 +81,6 @@ export function visibleWidth(str: string): number {
return width;
}
/**
* Extract ANSI escape sequences from a string at the given position.
*/
export function extractAnsiCode(str: string, pos: number): { code: string; length: number } | null {
if (pos >= str.length || str[pos] !== "\x1b") return null;
const next = str[pos + 1];
// CSI sequence: ESC [ ... m/G/K/H/J
if (next === "[") {
let j = pos + 2;
while (j < str.length && !/[mGKHJ]/.test(str[j]!)) j++;
if (j < str.length) return { code: str.substring(pos, j + 1), length: j + 1 - pos };
return null;
}
// OSC sequence: ESC ] ... BEL or ESC ] ... ST (ESC \)
// Used for hyperlinks (OSC 8), window titles, etc.
if (next === "]") {
let j = pos + 2;
while (j < str.length) {
if (str[j] === "\x07") return { code: str.substring(pos, j + 1), length: j + 1 - pos };
if (str[j] === "\x1b" && str[j + 1] === "\\") {
return { code: str.substring(pos, j + 2), length: j + 2 - pos };
}
j++;
}
return null;
}
return null;
}
const WRAP_OPTIONS = { wordWrap: true, hard: true, trim: false } as const;
/**