Use native width for non-ASCII TUI text

Keep the printable ASCII fast path, but send tabs, ANSI/control sequences, combining marks, and CJK through the same native width engine used by slicing and wrapping. This fixes Bun.stringWidth overcounting Arabic non-spacing marks and the old pure-ASCII tab double count.\n\nConstraint: Bun.stringWidth still reports Arabic combining marks as visible cells on Bun 1.3.14.\nRejected: Strip \p{Mn} in TypeScript | It would still leave tabs and future Unicode width mismatches divergent from native truncation.\nConfidence: high\nScope-risk: narrow\nDirective: Keep visibleWidth, sliceWithWidth, truncateToWidth, and wrapTextWithAnsi on one Unicode width model.\nTested: bun test packages/tui/test/text-utils.test.ts packages/tui/test/visible-width-jamo.test.ts packages/tui/test/wrap-ansi.test.ts packages/tui/test/truncate-to-width.test.ts packages/tui/test/render-regressions.test.ts; bun test packages/tui/test/*.test.ts; bun run check in packages/tui.\nNot-tested: Interactive terminal repaint with a live Arabic transcript.
This commit is contained in:
changhee.an
2026-05-29 16:44:31 +09:00
parent e707b90652
commit b9437a4276
2 changed files with 15 additions and 20 deletions
+8 -20
View File
@@ -4,6 +4,7 @@ import {
extractSegments as nativeExtractSegments,
sliceWithWidth as nativeSliceWithWidth,
truncateToWidth as nativeTruncateToWidth,
visibleWidth as nativeVisibleWidth,
wrapTextWithAnsi as nativeWrapTextWithAnsi,
type SliceResult,
} from "@oh-my-pi/pi-natives";
@@ -96,30 +97,17 @@ export function visibleWidthRaw(str: string): number {
return 0;
}
// Fast path: pure ASCII printable
let tabLength = 0;
const tabWidth = getDefaultTabWidth();
let isPureAscii = true;
let jamoOvercount = 0;
const isMacOS = process.platform === "darwin";
// Fast path: printable ASCII has one cell per code unit. Defer every
// control/non-ASCII case (tabs, ANSI/OSC, combining marks, CJK) to the
// native text engine so all width/slice/wrap helpers share one Unicode
// model instead of mixing Bun.stringWidth quirks with Rust truncation.
for (let i = 0; i < str.length; i++) {
const code = str.charCodeAt(i);
if (code === 9) {
tabLength += tabWidth;
} else if (code < 0x20 || code > 0x7e) {
isPureAscii = false;
// Hangul Compatibility Jamo (U+3131..U+318E) is EAW=W per UAX#11,
// but macOS terminals render them as 1 cell. WezTerm and others
// follow UAX#11 at 2 cells. Only correct on macOS.
if (isMacOS && code >= 0x3131 && code <= 0x318e) {
jamoOvercount++;
}
if (code < 0x20 || code > 0x7e) {
return nativeVisibleWidth(str, getDefaultTabWidth());
}
}
if (isPureAscii) {
return str.length + tabLength;
}
return Bun.stringWidth(str) - jamoOvercount + tabLength;
return str.length;
}
/**
+7
View File
@@ -7,6 +7,13 @@ describe("text utils", () => {
expect(visibleWidth(text)).toBe(2 + 3 + 5);
});
it("does not double-count pure ASCII tabs", () => {
expect(visibleWidth("a\tb")).toBe(1 + 3 + 1);
});
it("treats Arabic combining marks as zero-width", () => {
expect(visibleWidth("بَسِمَ")).toBe(3);
});
it("ignores OSC hyperlinks in visible width", () => {
const text = "\x1b]8;;https://example.com\x07link\x1b]8;;\x07";
expect(visibleWidth(text)).toBe(4);