Use native width for non-ASCII TUI text
Keep the printable ASCII fast path, but send tabs, ANSI/control sequences, combining marks, and CJK through the same native width engine used by slicing and wrapping. This fixes Bun.stringWidth overcounting Arabic non-spacing marks and the old pure-ASCII tab double count.\n\nConstraint: Bun.stringWidth still reports Arabic combining marks as visible cells on Bun 1.3.14.\nRejected: Strip \p{Mn} in TypeScript | It would still leave tabs and future Unicode width mismatches divergent from native truncation.\nConfidence: high\nScope-risk: narrow\nDirective: Keep visibleWidth, sliceWithWidth, truncateToWidth, and wrapTextWithAnsi on one Unicode width model.\nTested: bun test packages/tui/test/text-utils.test.ts packages/tui/test/visible-width-jamo.test.ts packages/tui/test/wrap-ansi.test.ts packages/tui/test/truncate-to-width.test.ts packages/tui/test/render-regressions.test.ts; bun test packages/tui/test/*.test.ts; bun run check in packages/tui.\nNot-tested: Interactive terminal repaint with a live Arabic transcript.
This commit is contained in:
@@ -4,6 +4,7 @@ import {
|
||||
extractSegments as nativeExtractSegments,
|
||||
sliceWithWidth as nativeSliceWithWidth,
|
||||
truncateToWidth as nativeTruncateToWidth,
|
||||
visibleWidth as nativeVisibleWidth,
|
||||
wrapTextWithAnsi as nativeWrapTextWithAnsi,
|
||||
type SliceResult,
|
||||
} from "@oh-my-pi/pi-natives";
|
||||
@@ -96,30 +97,17 @@ export function visibleWidthRaw(str: string): number {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Fast path: pure ASCII printable
|
||||
let tabLength = 0;
|
||||
const tabWidth = getDefaultTabWidth();
|
||||
let isPureAscii = true;
|
||||
let jamoOvercount = 0;
|
||||
const isMacOS = process.platform === "darwin";
|
||||
// Fast path: printable ASCII has one cell per code unit. Defer every
|
||||
// control/non-ASCII case (tabs, ANSI/OSC, combining marks, CJK) to the
|
||||
// native text engine so all width/slice/wrap helpers share one Unicode
|
||||
// model instead of mixing Bun.stringWidth quirks with Rust truncation.
|
||||
for (let i = 0; i < str.length; i++) {
|
||||
const code = str.charCodeAt(i);
|
||||
if (code === 9) {
|
||||
tabLength += tabWidth;
|
||||
} else if (code < 0x20 || code > 0x7e) {
|
||||
isPureAscii = false;
|
||||
// Hangul Compatibility Jamo (U+3131..U+318E) is EAW=W per UAX#11,
|
||||
// but macOS terminals render them as 1 cell. WezTerm and others
|
||||
// follow UAX#11 at 2 cells. Only correct on macOS.
|
||||
if (isMacOS && code >= 0x3131 && code <= 0x318e) {
|
||||
jamoOvercount++;
|
||||
}
|
||||
if (code < 0x20 || code > 0x7e) {
|
||||
return nativeVisibleWidth(str, getDefaultTabWidth());
|
||||
}
|
||||
}
|
||||
if (isPureAscii) {
|
||||
return str.length + tabLength;
|
||||
}
|
||||
return Bun.stringWidth(str) - jamoOvercount + tabLength;
|
||||
return str.length;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user