fix(tui): fixed visibleWidth parity and OSC66 width metadata behavior

- Fixed `visibleWidth` parity with native width using `Bun.stringWidth` ANSI stripping.
- Fixed tab and OSC66 width parsing with `s=`/`w=` metadata handling.
- Added ASCII and long-string fast paths in `visibleWidth` for faster width checks.
- Consolidated visible-width regression coverage by replacing removed IME/Arabic/Jamo tests with `visible-width.test.ts`.
This commit is contained in:
can1357
2026-06-07 06:24:07 +02:00
parent a78f4c6766
commit 705c937c56
9 changed files with 174 additions and 441 deletions
+1 -1
View File
@@ -20,7 +20,7 @@
- Changed the plan-mode active prompt (`prompts/system/plan-mode-active.md`) to make plans decision-complete and cut filler. Added an Objective framing ("another engineer can execute end-to-end without making a single design decision"), a shared "Resolving Unknowns" section (explore discoverable facts before asking; reserve `ask` for non-derivable preferences/tradeoffs with 2–4 options + a recommended default), and a single shared "The Plan" structure (Context / Approach grouped by behavior not file-by-file / ≤5 Critical files / Verification / Assumptions) that replaces the per-branch structure guidance previously duplicated across the iterative and parallel workflows. Added explicit prohibitions on sections that decide nothing (Non-Goals, Out of Scope, Alternatives Considered, Risks/Mitigations boilerplate, Future Work), on enumerating every file/line, and on inventing schema/validation/precedence policy the request never established.
- Changed completion notifications (`completion.notify`) to fire whenever the agent yields its turn, including in the foreground. The `agent_end` notification was previously gated behind background mode (`isBackgrounded`), so an ordinary foreground turn never emitted one; the gate is gone and the desktop toast now fires on every normal turn completion (still skipped for aborted/error turns and when `completion.notify` is `off`).
- Changed the in-progress `task` tool block to keep the shared `context` brief (`# Goal` / `# Constraints` background) visible after the first progress snapshot arrives, instead of dropping it the moment the streaming call view was replaced by the result frame, and to stop animating a spinner/clock next to the `Task` frame header while running — the per-agent body lines already carry their own running spinner, so the header now shows a static state icon (matching the completed/failed header icons). The context is rendered through a shared `buildContextSection` helper that also undoes per-field double-encoding, so the brief reads cleanly in the result frame even though `renderResult` receives the raw (un-repaired) tool args.
- Changed the label shown when you press Esc to interrupt a streaming turn from the ambiguous `Operation aborted` to `Interrupted by user`, so a deliberate user interrupt no longer reads like an internal failure. The Esc handler threads the reason through `AgentSession.abort({ reason })` → `Agent.abort(reason)` so it rides the `AbortController` onto the aborted assistant message's `errorMessage` and renders verbatim on both the live and replay paths; aborts that carry no reason still fall back to the retry-aware `Operation aborted` generic. The transcript label resolution is centralized in `resolveAbortLabel` (`session/messages.ts`).
- Changed the messaging shown when you press Esc to interrupt a streaming turn from the ambiguous `Operation aborted` / `Tool execution was aborted: Request was aborted` to `Interrupted by user`, so a deliberate user interrupt no longer reads like an internal failure. Every Esc/flush interrupt path (`onEscape` while streaming, the queued-message restore-and-abort path, and the empty-submit queue flush) threads the reason through `AgentSession.abort({ reason })` → `Agent.abort(reason)` so it rides the `AbortController` onto the aborted assistant message's `errorMessage`; the turn label renders it verbatim on both the live and replay paths, and the synthetic placeholder results paired with in-flight tool calls now read `Tool execution was aborted: Interrupted by user`. Aborts that carry no reason still fall back to the retry-aware `Operation aborted` generic. Transcript label resolution is centralized in `resolveAbortLabel` (`session/messages.ts`).
### Fixed
+3
View File
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Removed Kitty temp-file image transmission, its startup support probe, the `PI_KITTY_IMAGE_TRANSMISSION` override, and the temp-file helper exports. Kitty/Ghostty image payloads now stay on in-band base64 before placeholder/direct placement, avoiding blank first renders from temp-file load races.
@@ -23,6 +24,8 @@
### Fixed
- Fixed `visibleWidth()` so terminal column measurements for ANSI and OSC text now match the native truncation/wrapping helpers, including OSC 66 text-sizing spans being counted at their scaled payload width
- Fixed cursor, padding, and line-fit behavior when strings contain tabs or OSC escapes by aligning `visibleWidth()` with the native text-width model
- Fixed the transcript — or a re-appearing prior view such as the welcome screen — duplicating itself on terminals without a scroll-position oracle (Ghostty/kitty/iTerm/WezTerm) when a foreground tool completes by rewriting a partly-committed block, or when the transcript is reset. A non-destructive viewport repaint no longer re-paints rows that are byte-identical to what is already committed to native scrollback into the active grid; the repaint anchor is clamped to the committed-and-unchanged prefix (`min(firstChanged, scrollbackHighWater)`).
## [15.10.0] - 2026-06-06
+1 -1
View File
@@ -5,7 +5,7 @@
*/
import { visibleWidth as nativeVisibleWidth } from "@oh-my-pi/pi-natives";
import { getDefaultTabWidth } from "@oh-my-pi/pi-utils";
import { visibleWidthRaw as hybridVisibleWidth, replaceTabs } from "../src/utils";
import { visibleWidth as hybridVisibleWidth, replaceTabs } from "../src/utils";
const ITERATIONS = 10_000;
const WARMUP = 500;
+92 -59
View File
@@ -4,7 +4,6 @@ import {
extractSegments as nativeExtractSegments,
sliceWithWidth as nativeSliceWithWidth,
truncateToWidth as nativeTruncateToWidth,
visibleWidth as nativeVisibleWidth,
wrapTextWithAnsi as nativeWrapTextWithAnsi,
type SliceResult,
} from "@oh-my-pi/pi-natives";
@@ -145,44 +144,6 @@ export function padding(n: number): string {
// Grapheme segmenter (shared instance)
const segmenter = new Intl.Segmenter(undefined, { granularity: "grapheme" });
const EXTENDED_PICTOGRAPHIC_REGEX = /\p{Extended_Pictographic}/u;
// Matches CSI (`\x1b[…`) and OSC (`\x1b]…` terminated by BEL/ST) escape
// sequences. Mirrors the standard ansi-regex coverage so visible-span
// segmentation lines up with the native ANSI scanner.
const ANSI_ESCAPE_REGEX =
/[\u001b\u009b][[\]()#;?]*(?:(?:(?:(?:;[-a-zA-Z\d/#&.:=?%@~_]+)*|[a-zA-Z\d]+(?:;[-a-zA-Z\d/#&.:=?%@~_]*)*)?\u0007)|(?:(?:\d{1,4}(?:;\d{0,4})*)?[\dA-PR-TZcf-nq-uy=><~]))/g;
function pictographicSpanWidth(span: string): number {
let width = 0;
for (const { segment } of segmenter.segment(span)) {
width += EXTENDED_PICTOGRAPHIC_REGEX.test(segment) ? 2 : nativeVisibleWidth(segment, getDefaultTabWidth());
}
return width;
}
// Width fallback for strings that mix ANSI styling with ZWJ pictographic
// emoji. `Intl.Segmenter` would split an escape sequence into individual
// graphemes, so the native scanner (which only skips ANSI when handed the
// complete sequence) double-counts the printable SGR bytes. Excise the ANSI
// spans first — they contribute zero cells — and apply the pictographic
// grapheme override only to the visible spans, then sum.
function visibleWidthByGrapheme(str: string): number {
let width = 0;
let lastIndex = 0;
ANSI_ESCAPE_REGEX.lastIndex = 0;
for (let match = ANSI_ESCAPE_REGEX.exec(str); match !== null; match = ANSI_ESCAPE_REGEX.exec(str)) {
if (match.index > lastIndex) {
width += pictographicSpanWidth(str.slice(lastIndex, match.index));
}
lastIndex = ANSI_ESCAPE_REGEX.lastIndex;
}
if (lastIndex < str.length) {
width += lastIndex === 0 ? pictographicSpanWidth(str) : pictographicSpanWidth(str.slice(lastIndex));
}
return width;
}
/**
* Get the shared grapheme segmenter instance.
*/
@@ -190,32 +151,104 @@ export function getSegmenter(): Intl.Segmenter {
return segmenter;
}
export function visibleWidthRaw(str: string): number {
if (!str) {
return 0;
}
// Kitty OSC 66 text-sizing spans: `\x1b]66;<meta>;<payload>` terminated by BEL
// or ST. `Bun.stringWidth` strips the whole span (payload included) to zero
// cells, but the payload is visible and scales by the `s=` factor, so each is
// added back so width matches the native truncate/slice/wrap helpers.
const OSC66_SPAN_REGEX = /\x1b\]66;([^;]*);([\s\S]*?)(?:\x07|\x1b\\)/g;
const OSC66_PREFIX = "\x1b]66;";
const ESC = "\x1b";
const TAB = "\t";
const LONG_WIDTH_FAST_PATH_MIN = 128;
// Fast path: printable ASCII has one cell per code unit. Defer every
// control/non-ASCII case (tabs, ANSI/OSC, combining marks, CJK) to the
// native text engine so all width/slice/wrap helpers share one Unicode
// model instead of mixing Bun.stringWidth quirks with Rust truncation.
for (let i = 0; i < str.length; i++) {
const code = str.charCodeAt(i);
if (code < 0x20 || code > 0x7e) {
const tabWidth = getDefaultTabWidth();
if (str.includes("\x1b]66;")) return nativeVisibleWidth(str, tabWidth);
return str.includes("\u200d") ? visibleWidthByGrapheme(str) : nativeVisibleWidth(str, tabWidth);
}
}
return str.length;
}
// Pin Bun.stringWidth semantics to the native width engine and guard against Bun
// default drift: strip ANSI/OSC (don't count escape bytes) and treat
// ambiguous-width East Asian chars as narrow (1 cell), matching `unicode-width`'s
// non-CJK tables that back truncate/slice/wrap. Hoisted so no per-call alloc.
const STRING_WIDTH_OPTS = { countAnsiEscapeCodes: false, ambiguousIsNarrow: true } as const;
/**
* Calculate the visible width of a string in terminal columns.
* Visible width of a string in terminal columns, excluding ANSI/OSC escapes.
*
* `Bun.stringWidth` does the heavy lifting (UAX#11 width tables + ANSI/OSC
* stripping); this adds the two corrections it omits — tabs (expanded to
* `tabWidth` cells) and OSC 66 text-sizing payloads (scaled by `s=`).
*/
export function visibleWidth(str: string): number {
if (!str) return 0;
return visibleWidthRaw(str);
// Long non-escape text is faster through Bun's native scanner than through
// a JS printable-ASCII prepass. Escape-bearing strings stay on the scanner
// below so CSI/OSC-heavy render output can still bail out at the first ESC.
if (str.length >= LONG_WIDTH_FAST_PATH_MIN && !str.includes(ESC)) {
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
let tabCount = 0;
for (let tabIndex = str.indexOf(TAB); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
}
if (tabCount > 0) width += tabCount * getDefaultTabWidth();
return width;
}
let tabCount = 0;
let i = 0;
for (; i < str.length; i++) {
const code = str.charCodeAt(i);
if (code < 0x20 || code > 0x7e) {
if (code === 0x09) {
tabCount++;
continue;
}
break;
}
}
if (i === str.length) {
return tabCount === 0 ? str.length : str.length + tabCount * (getDefaultTabWidth() - 1);
}
if (tabCount === 0) {
let tabIndex = str.indexOf(TAB, i + 1);
if (tabIndex !== -1) {
tabCount = 1;
for (tabIndex = str.indexOf(TAB, tabIndex + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
}
}
} else {
for (let tabIndex = str.indexOf(TAB, i + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
}
}
// `Bun.stringWidth` is a JSC builtin (no per-call N-API number box, unlike
// the native scanner that traps under Bun 1.3.x GC/N-API load). It strips
// CSI/OSC to zero cells and shares the native engine's UAX#11 width tables.
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
if (tabCount > 0) width += tabCount * getDefaultTabWidth();
// OSC 66: add back each stripped span as `scale * (explicit w ?? payload
// width)`. Matched rather than replaced to avoid reallocating the string.
if (str.includes(OSC66_PREFIX, i)) {
OSC66_SPAN_REGEX.lastIndex = 0;
for (let m = OSC66_SPAN_REGEX.exec(str); m !== null; m = OSC66_SPAN_REGEX.exec(str)) {
let scale = 1;
let explicit: number | undefined;
for (const part of m[1].split(":")) {
// metadata keys are single chars, e.g. `s=2`, `w=5`
if (part.indexOf("=") !== 1) continue;
const value = Number.parseInt(part.slice(2), 10);
if (!Number.isFinite(value)) continue;
if (part[0] === "s") {
if (value >= 1 && value <= 7) scale = value;
} else if (part[0] === "w" && value > 0) {
explicit = value;
}
}
width += scale * (explicit ?? Bun.stringWidth(m[2], STRING_WIDTH_OPTS));
}
}
return width;
}
const THAI_LAO_AM_REGEX = /[\u0e33\u0eb3]/;
-109
View File
@@ -1,109 +0,0 @@
/**
* Regression guard for Korean IME cursor positioning in the Input
* component.
*
* Background: macOS Korean IME (2-bul keyboard) emits Hangul Compatibility
* Jamo (U+3131..U+318E) during composition. Before commit 79e3170c6 the
* Input.render() computed `cursorCols = visibleWidth(value.slice(0,
* cursorIndex))` and `Bun.stringWidth` returned 2 for each jamo (per UAX
* #11 EAW=W), while every macOS terminal renders them as 1 cell.
*
* The user-visible bug was a GROWING horizontal gap between the typed
* jamo and the IME candidate window — every additional jamo doubled the
* offset because `cursorCols` was N×2 instead of N×1. With 14 jamo typed,
* the gap was ~14 cells.
*
* After the JS-side width correction, `cursorCols` is N×1. The Rust
* `pi-natives` `sliceWithWidth` still treats jamo as 2 cells (binary
* package; follow-up), so the cursor marker placement in `Input.render()`
* has a residual ≤1-cell-per-jamo offset — but the user-visible "growing
* gap" is gone because the JS-side `cursorCols` no longer doubles.
*
* What this test guards:
* - The cursor column for a value containing N jamo is **bounded above
* by `PROMPT_WIDTH + N`** (the correct cell count after the fix).
* Before the fix it would have been ~`PROMPT_WIDTH + 2N`, which is
* what the test catches.
* - Pure-ASCII and Hangul-syllable baselines are unchanged.
*
* What this test does NOT guard:
* - The Rust-side `sliceWithWidth` jamo discrepancy. Tracked as a
* follow-up; will tighten this test once the Rust crate is rebuilt.
*/
import { describe, expect, it } from "bun:test";
import { CURSOR_MARKER } from "@oh-my-pi/pi-tui";
import { Input } from "@oh-my-pi/pi-tui/components/input";
import { visibleWidth } from "@oh-my-pi/pi-tui/utils";
/**
* Drive `text` through `Input.handleInput()` one Unicode code point at a
* time (mirrors what the IME does — one code point per emitted sequence),
* then return the visual column where the hardware cursor marker lands
* in the rendered output.
*/
function cursorColAfterTyping(text: string, width = 80): number {
const input = new Input();
(input as unknown as { focused: boolean }).focused = true;
for (const char of text) {
input.handleInput(char);
}
const [line] = input.render(width);
const markerIdx = line.indexOf(CURSOR_MARKER);
if (markerIdx < 0) {
throw new Error(`CURSOR_MARKER not found in rendered line: ${JSON.stringify(line)}`);
}
return visibleWidth(line.slice(0, markerIdx));
}
const PROMPT_WIDTH = 2; // "> "
// The jamo-cursor regression (PR #1410 / origin issue) only applies on
// macOS, where terminals render Hangul Compatibility Jamo as 1 cell while
// UAX#11 (and `Bun.stringWidth`) report 2. Off-darwin both the terminal and
// the width helper agree on 2, so the doubling regression cannot occur.
const IS_DARWIN = process.platform === "darwin";
describe("Input cursor column does not grow at 2× per jamo", () => {
it("ASCII baseline: cursor lands exactly after the typed text", () => {
expect(cursorColAfterTyping("hello")).toBe(PROMPT_WIDTH + 5);
});
it("Hangul syllables: cursor lands exactly after typed text (2 cells each)", () => {
expect(cursorColAfterTyping("안녕")).toBe(PROMPT_WIDTH + 4);
});
it.skipIf(!IS_DARWIN)("single jamo: cursor column is at most `PROMPT_WIDTH + 1`", () => {
// Before fix: PROMPT_WIDTH + 2 = 4. After fix: ≤ 3.
expect(cursorColAfterTyping("ㅁ")).toBeLessThanOrEqual(PROMPT_WIDTH + 1);
});
it.skipIf(!IS_DARWIN)("8 consecutive jamo: cursor column is at most `PROMPT_WIDTH + 8`", () => {
// Before fix: PROMPT_WIDTH + 16 = 18. After fix: ≤ 10.
expect(cursorColAfterTyping("ㅁㄴㅁㄴㅇㅂㄴㅂ")).toBeLessThanOrEqual(PROMPT_WIDTH + 8);
});
it.skipIf(!IS_DARWIN)("20 consecutive jamo: cursor column is at most `PROMPT_WIDTH + 20`", () => {
// Before fix: PROMPT_WIDTH + 40 = 42 (catastrophic gap). After fix: ≤ 22.
const jamo = "ㅁㄴㄷㅂㅈㅎㅋㅌㄱㄹ".repeat(2);
expect(cursorColAfterTyping(jamo)).toBeLessThanOrEqual(PROMPT_WIDTH + 20);
});
it.skipIf(!IS_DARWIN)("cursor column grows by ≤1 per typed jamo (not 2)", () => {
// The regression: each typed jamo would advance cursor by 2 columns
// instead of 1, doubling the offset every keystroke. Assert the
// per-step delta never exceeds 1.
const input = new Input();
(input as unknown as { focused: boolean }).focused = true;
const jamo = "ㅁㄴㅇㅂㅈㅎㅋㅌㄷㄹ";
let prevCol = PROMPT_WIDTH;
for (let i = 0; i < jamo.length; i++) {
input.handleInput(jamo[i]);
const [line] = input.render(80);
const markerIdx = line.indexOf(CURSOR_MARKER);
const col = visibleWidth(line.slice(0, markerIdx));
const delta = col - prevCol;
expect(delta).toBeLessThanOrEqual(1);
prevCol = col;
}
});
});
-122
View File
@@ -1,122 +0,0 @@
import { describe, expect, it } from "bun:test";
import { type Component, TUI, visibleWidth } from "@oh-my-pi/pi-tui";
import { VirtualTerminal } from "./virtual-terminal";
// Regression test for https://github.com/can1357/oh-my-pi/issues/643
//
// Arabic text with tashkeel (nonspacing diacritics, Unicode category Mn) was
// over-measured: each mark counted as 1 column instead of 0. A line of Quranic
// text with N marks therefore reported a width N columns larger than what the
// terminal actually renders, and the renderer crashed with "Rendered line
// exceeds terminal width" (13.19.0) once enough marks accumulated.
//
// Root cause: `Bun.stringWidth` counts Mn marks as 1 column (still true as of
// Bun 1.3.x). The fix routes every non-ASCII line through the native width
// engine (Rust `unicode-width`), which measures nonspacing marks as 0 — in
// agreement with xterm.js (BMP_COMBINING tables) and real terminals.
//
// The contract defended here: tashkeel marks contribute zero columns, so a
// marked-up line whose letter width exactly equals the terminal width renders
// unclipped on a single row. Over-measurement would either truncate trailing
// letters (today's fitting behavior) or crash (the original report).
// Tashkeel samples from the issue, with their correct letter-cell widths.
const TASHKEEL_SAMPLES: ReadonlyArray<readonly [text: string, width: number]> = [
["بِسْمِ", 3], // 3 letters + 3 marks
["سَابِقُوا", 6], // 6 letters + 3 marks
["فَتَبَيَّنُوا", 7], // 7 letters + 5 marks (incl. shadda)
];
class RawLinesComponent implements Component {
#lines: string[];
constructor(lines: string[]) {
this.#lines = [...lines];
}
setLines(lines: string[]): void {
this.#lines = [...lines];
}
invalidate(): void {}
render(): string[] {
return [...this.#lines];
}
}
async function settle(term: VirtualTerminal): Promise<void> {
const nextTick = Promise.withResolvers<void>();
process.nextTick(nextTick.resolve);
await nextTick.promise;
await Bun.sleep(20);
await term.flush();
}
describe("issue #643: Arabic tashkeel width measurement", () => {
it("measures nonspacing tashkeel marks as zero columns", () => {
for (const [text, width] of TASHKEEL_SAMPLES) {
expect(visibleWidth(text)).toBe(width);
}
// Mixed ASCII + tashkeel stays on the non-ASCII measurement path.
expect(visibleWidth(`ayah: ${TASHKEEL_SAMPLES[1][0]}`)).toBe(6 + TASHKEEL_SAMPLES[1][1]);
});
it("renders a full-width tashkeel line unclipped on a single terminal row", async () => {
// "quran: " (7 cells) + 6-cell word = 13 cells — an exact fit at width 13.
// The original bug measured the word as 9 (6 letters + 3 marks), making the
// line appear 3 columns too wide, so the renderer's width fitting would
// truncate real trailing letters before writing.
const word = TASHKEEL_SAMPLES[1][0];
const line = `quran: ${word}`;
const width = 13;
expect(visibleWidth(line)).toBe(width);
const term = new VirtualTerminal(width, 6);
const tui = new TUI(term);
const component = new RawLinesComponent(["header", line, "tail"]);
tui.addChild(component);
try {
tui.start();
await settle(term);
const viewport = term.getViewport().map(row => row.trimEnd());
// The full marked-up text survives the round trip: no truncation by the
// renderer, no clipping or wrapping by the terminal.
expect(viewport[1]).toBe(line);
expect(viewport[0]).toBe("header");
expect(viewport[2]).toBe("tail");
// Exactly one terminal row per logical row — the marks did not push the
// line across the right margin.
expect(term.getScrollBuffer().length).toBe(6);
} finally {
tui.stop();
}
});
it("keeps row accounting exact when tashkeel rows scroll into native history", async () => {
const word = TASHKEEL_SAMPLES[2][0];
const width = 24;
const height = 5;
const term = new VirtualTerminal(width, height);
const tui = new TUI(term);
const tashkeelRows = Array.from({ length: 8 }, (_v, i) => `ayah-${i} ${word}`);
const component = new RawLinesComponent(tashkeelRows);
tui.addChild(component);
try {
tui.start();
await settle(term);
// 8 content rows at height 5 → 3 rows pushed into scrollback.
const buffer = term.getScrollBuffer().map(row => row.trimEnd());
expect(buffer.length).toBe(tashkeelRows.length);
for (let i = 0; i < tashkeelRows.length; i++) {
expect(buffer[i]).toBe(tashkeelRows[i]);
}
} finally {
tui.stop();
}
});
});
-3
View File
@@ -17,9 +17,6 @@ describe("text utils", () => {
expect(visibleWidth("a\tb")).toBe(1 + 3 + 1);
});
it("treats Arabic combining marks as zero-width", () => {
expect(visibleWidth("بَسِمَ")).toBe(3);
});
it("ignores OSC hyperlinks in visible width", () => {
const text = "\x1b]8;;https://example.com\x07link\x1b]8;;\x07";
expect(visibleWidth(text)).toBe(4);
@@ -1,146 +0,0 @@
/**
* Regression guard for the Hangul Compatibility Jamo width correction in
* `visibleWidthRaw`.
*
* `Bun.stringWidth` (and the underlying UAX#11 EAW tables) classify Hangul
* Compatibility Jamo (U+3131..U+318E) as Wide (2 cells), but every macOS
* terminal we ship to (Ghostty, Terminal.app, iTerm2) actually renders them
* as a single cell. Without the correction, `#extractCursorPosition` doubles
* the column count for every jamo emitted by a Korean IME during
* composition, displacing the hardware cursor (and therefore the IME
* candidate window) `N_jamo` cells past the actual glyph.
*
* Hangul Syllables (U+AC00..U+D7A3, e.g. `안`) are correctly 2 cells in both
* Bun and the terminal — make sure the fix did NOT regress that. The
* Halfwidth Hangul block (U+FFA0..U+FFDC) is already classified as Narrow
* by Bun, so it does not appear in the correction and the test below is a
* regression sanity check.
*/
import { describe, expect, it } from "bun:test";
import { Ellipsis, sliceWithWidth, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui/utils";
// The macOS-only correction (see PR #1410) keeps jamo at 1 cell on darwin;
// every other platform follows UAX#11 and reports 2 cells per jamo.
const JAMO_CELLS = process.platform === "darwin" ? 1 : 2;
describe("visibleWidth — Hangul Compatibility Jamo correction", () => {
it("single compatibility jamo is 1 cell on darwin, 2 elsewhere", () => {
// U+3141 HANGUL LETTER MIEUM
expect(visibleWidth("ㅁ")).toBe(JAMO_CELLS);
// U+3134 HANGUL LETTER NIEUN
expect(visibleWidth("ㄴ")).toBe(JAMO_CELLS);
// U+3147 HANGUL LETTER IEUNG
expect(visibleWidth("ㅇ")).toBe(JAMO_CELLS);
// U+3142 HANGUL LETTER PIEUP
expect(visibleWidth("ㅂ")).toBe(JAMO_CELLS);
// U+3148 HANGUL LETTER JIEUJ
expect(visibleWidth("ㅈ")).toBe(JAMO_CELLS);
});
it("range edges U+3131 and U+318E follow platform width", () => {
// U+3131 HANGUL LETTER KIYEOK — first jamo in the block
expect(visibleWidth("\u3131")).toBe(JAMO_CELLS);
// U+318E HANGUL LETTER ARAEAE — last jamo in the block
expect(visibleWidth("\u318e")).toBe(JAMO_CELLS);
});
it("U+3164 HANGUL FILLER (inside the corrected range) follows platform width", () => {
// Often emitted by IME for empty-syllable placeholders. The filler is the
// one code point in the block that UAX#11 / `unicode-width` classify as
// zero-width, so off-darwin it measures 0 cells. On darwin the blanket
// jamo correction (U+3131..U+318E → 1) forces it to a single cell.
const fillerCells = process.platform === "darwin" ? 1 : 0;
expect(visibleWidth("\u3164")).toBe(fillerCells);
});
it("combining marks on compatibility jamo keep the platform base width", () => {
// Combining marks add no cells. The native scanner must keep
// UnicodeWidthStr's sequence rules, then apply the same local jamo
// correction used for standalone code points.
expect(visibleWidth("\u3141\u0301")).toBe(JAMO_CELLS);
const fillerCells = process.platform === "darwin" ? 1 : 0;
expect(visibleWidth("\u3164\u0301")).toBe(fillerCells);
});
it("string of 8 consecutive jamo is 8 cells on darwin, 16 elsewhere", () => {
// Matches the user-typed sequence in the v2 screen recording —
// before the macOS fix this returned 16 and produced an 8-cell gap.
expect(visibleWidth("ㅁㄴㅁㄴㅇㅂㄴㅂ")).toBe(8 * JAMO_CELLS);
});
it("Hangul Syllables (U+AC00..U+D7A3) stay at 2 cells", () => {
// `안` U+C548 — composed syllable, must remain 2 cells
expect(visibleWidth("안")).toBe(2);
// `녕` U+B155 — composed syllable
expect(visibleWidth("녕")).toBe(2);
// Whole word: 안녕 = 4 cells
expect(visibleWidth("안녕")).toBe(4);
// First & last in the block, for boundary coverage
expect(visibleWidth("\uac00")).toBe(2); // 가
expect(visibleWidth("\ud7a3")).toBe(2); // 힣
});
it("mixed ASCII + syllable + jamo strings add correctly", () => {
// a (1) + 안 (2) + ㅂ (J) + b (1) = 4 + J
expect(visibleWidth("a안ㅂb")).toBe(4 + JAMO_CELLS);
// 11 ASCII letters + 1 syllable + 4 jamo = 11 + 2 + 4*J
expect(visibleWidth("hello world안ㅁㄴㅇㅂ")).toBe(11 + 2 + 4 * JAMO_CELLS);
});
it("does not regress ASCII fast path or empty input", () => {
expect(visibleWidth("")).toBe(0);
expect(visibleWidth("hello")).toBe(5);
expect(visibleWidth("a")).toBe(1);
// Tab character (ASCII 0x09) inside the fast path expands to >2
expect(visibleWidth("a\tb")).toBeGreaterThan(2);
});
it("does not change width for other CJK characters", () => {
// Chinese: 漢字 (each 2 cells)
expect(visibleWidth("漢字")).toBe(4);
// Japanese hiragana: あい (each 2 cells)
expect(visibleWidth("あい")).toBe(4);
// Japanese katakana: アイ (each 2 cells)
expect(visibleWidth("アイ")).toBe(4);
});
it("Halfwidth Hangul block is unaffected (already Narrow in Bun)", () => {
// U+FFA1 HALFWIDTH HANGUL LETTER KIYEOK — Bun reports 1, untouched.
expect(visibleWidth("\uffa1")).toBe(1);
// U+FFDC HALFWIDTH HANGUL LETTER I
expect(visibleWidth("\uffdc")).toBe(1);
});
});
describe("native text helpers — Hangul Compatibility Jamo correction", () => {
// These exercise the Rust-side `char_width_corrected` wrapper in
// crates/pi-natives/src/text.rs. They will fail until the native
// binding is rebuilt (`bun run build:native`); CI rebuilds natives so
// they pass there. Mirrors the TS-side range U+3131..=U+318E.
it("sliceWithWidth treats jamo per platform width", () => {
// 8 jamo at JAMO_CELLS cells each must fit fully within the
// platform's natural cell count.
const input = "ㅁ".repeat(8);
const { text, width } = sliceWithWidth(input, 0, 8 * JAMO_CELLS, true);
expect(text).toBe(input);
expect(width).toBe(8 * JAMO_CELLS);
});
it("truncateToWidth keeps 8 jamo within their platform budget", () => {
// On darwin (1 cell/jamo) 8 jamo fit in 8 cells; off-darwin they need 16.
const result = truncateToWidth("ㅁ".repeat(20), 8 * JAMO_CELLS, Ellipsis.Omit);
// Strip any trailing pad to count jamo content.
const jamo = result.replaceAll(/[^\u3131-\u318E]/g, "");
expect(jamo.length).toBe(8);
});
it("native and TS visibleWidth agree on a jamo run", () => {
// Cross-layer parity guard: without the native fix, the TS path
// (Bun.stringWidth + manual correction) and the native path
// (unicode_width) disagreed by a factor of 2.
const input = "ㅁㄴㅇㅂㅈㄷㄱㅅ";
expect(visibleWidth(input)).toBe(8 * JAMO_CELLS);
expect(sliceWithWidth(input, 0, 8 * JAMO_CELLS, true).width).toBe(8 * JAMO_CELLS);
});
});
+77
View File
@@ -0,0 +1,77 @@
/**
* `visibleWidth` measures terminal column width via `Bun.stringWidth` (a JSC
* builtin) instead of the native scanner, to keep the render loop off the
* N-API number-boxing path that traps under Bun 1.3.x GC pressure.
*
* Correctness contract: the result MUST equal the native engine's width for the
* same input, because `truncateToWidth` / `sliceWithWidth` / `wrapTextWithAnsi`
* cut text using that native model — any divergence makes padding / cursor math
* (`width - visibleWidth(...)`) drift. This guards the two corrections layered
* on top of `Bun.stringWidth` (tabs, OSC 66 scaling) and catches silent
* `Bun.stringWidth` width-table drift across Bun upgrades.
*/
import { describe, expect, it } from "bun:test";
import { visibleWidth as nativeVisibleWidth } from "@oh-my-pi/pi-natives";
import { getDefaultTabWidth, visibleWidth } from "@oh-my-pi/pi-tui/utils";
const ESC = "\x1b";
const ST = "\x1b\\";
const BEL = "\x07";
const TAB = getDefaultTabWidth();
describe("visibleWidth — parity with the native width engine", () => {
const corpus: [string, string][] = [
["empty", ""],
["ascii", "Pending run: passed"],
["styled", `${ESC}[31mred${ESC}[0m text`],
["styled-truecolor", `${ESC}[38;2;1;2;3mx${ESC}[0m`],
["nested-sgr", `${ESC}[1m${ESC}[31mbold${ESC}[0m${ESC}[0m`],
["osc8-st", `${ESC}]8;;https://x.com${ST}link${ESC}]8;;${ST}`],
["osc8-bel", `${ESC}]8;;u${BEL}t${ESC}]8;;${BEL}`],
["cjk", "日本語のテキスト"],
["cjk-mixed", "abc中文def"],
["hangul-syllables", "안녕하세요"],
["styled-cjk", `${ESC}[1m漢字${ESC}[0m`],
["emoji", "👍 done"],
["emoji-zwj", "👨‍👩‍👧‍👦"],
["emoji-flag", "🇯🇵"],
["styled-zwj", `${ESC}[31m👨‍👩‍👧‍👦${ESC}[0m`],
["variation-selector", "▶️"],
["combining", "e\u0301"],
["ambiguous", "§±×→①②③"],
["box-drawing", "─│┌┐└┘"],
["fullwidth", "123"],
["halfwidth-kana", "アイウ"],
["rtl-arabic", "مرحبا"],
["thai", "สวัสดี"],
["tabs", "name\tvalue\tstatus"],
["leading-tabs", "\t\tindented"],
["osc66-scale", `${ESC}]66;s=2;big${ST}`],
["osc66-explicit-w", `${ESC}]66;w=5;Hi${BEL}`],
["osc66-scale-and-w", `${ESC}]66;s=3:w=4;X${ST}`],
["osc66-cjk", `${ESC}]66;s=2;日本${ST}`],
["osc66-inline", `pre ${ESC}]66;s=2;AB${ST} post`],
["osc66-multi", `${ESC}]66;s=2;A${ST} ${ESC}]66;s=3;B${ST}`],
["osc66-with-tabs", `\t${ESC}]66;s=2;X${ST}\t`],
];
for (const [name, input] of corpus) {
it(name, () => {
expect(visibleWidth(input)).toBe(nativeVisibleWidth(input, TAB));
});
}
it("strips ANSI (styled text measures as its plain content)", () => {
expect(visibleWidth(`${ESC}[31mhello${ESC}[0m`)).toBe(5);
});
it("expands each tab to the configured tab width", () => {
expect(visibleWidth("a\tb")).toBe(2 + TAB);
expect(visibleWidth("\t\t")).toBe(2 * TAB);
});
it("scales OSC 66 text-sizing payloads by `s=`", () => {
expect(visibleWidth(`${ESC}]66;s=2;big${ST}`)).toBe(6); // 2 * width("big")
expect(visibleWidth(`${ESC}]66;w=5;Hi${BEL}`)).toBe(5); // explicit width, scale 1
expect(visibleWidth(`${ESC}]66;s=3:w=4;X${ST}`)).toBe(12); // 3 * 4
});
});