feat(text): refactored text processing with UTF-16 optimization and ANSI state bitflags
- Added `visibleWidth()` function to measure visible text width while excluding ANSI codes. - Optimized text processing from UTF-8 to UTF-16 implementation with direct JsString handling for improved performance. - Refactored ANSI state tracking to use bitflags-based SgrState struct instead of individual boolean fields for more efficient storage. - Introduced ColorCode enum to represent color codes (Basic, Indexed, Rgb) for better type safety. - Added comprehensive test coverage for UTF-16 text processing, ANSI detection, and width calculations. - Updated napi feature from 'napi8' to 'napi10' and added unicode-segmentation dependency.
This commit is contained in:
Generated
+1
@@ -805,6 +805,7 @@ dependencies = [
|
||||
"napi-derive",
|
||||
"rayon",
|
||||
"syntect",
|
||||
"unicode-segmentation",
|
||||
"unicode-width",
|
||||
]
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ crate-type = ["cdylib"]
|
||||
workspace = true
|
||||
|
||||
[dependencies]
|
||||
napi = { version = "3", features = ["napi8", "tokio_rt"] }
|
||||
napi = { version = "3", features = ["napi10", "tokio_rt"] }
|
||||
napi-derive = "3"
|
||||
grep-regex = "0.1"
|
||||
grep-searcher = "0.1"
|
||||
@@ -28,6 +28,7 @@ image = { version = "0.25", default-features = false, features = [
|
||||
"webp",
|
||||
] }
|
||||
bstr = "1"
|
||||
unicode-segmentation = "1.11"
|
||||
unicode-width = "0.2"
|
||||
syntect = { version = "5.3", default-features = false, features = [
|
||||
"default-syntaxes",
|
||||
|
||||
+760
-415
File diff suppressed because it is too large
Load Diff
@@ -135,7 +135,7 @@ function resolveModelOverride(
|
||||
): { model?: Model<Api>; thinkingLevel?: ThinkingLevel } {
|
||||
if (modelPatterns.length === 0) return {};
|
||||
const matchPreferences = { usageOrder: settings?.getStorage()?.getModelUsageOrder() };
|
||||
const roles = settings?.serialize().modelRoles as Record<string, string> | undefined;
|
||||
const roles = settings?.getGroup("modelRoles");
|
||||
for (const pattern of modelPatterns) {
|
||||
const normalized = pattern.trim().toLowerCase();
|
||||
if (!normalized || DEFAULT_MODEL_ALIASES.has(normalized)) {
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added `visibleWidth()` function to measure the visible width of text, excluding ANSI codes
|
||||
|
||||
## [9.6.1] - 2026-02-01
|
||||
### Added
|
||||
|
||||
@@ -75,6 +75,7 @@ export {
|
||||
type SliceWithWidthResult,
|
||||
sliceWithWidth,
|
||||
truncateToWidth,
|
||||
visibleWidth,
|
||||
} from "./text/index";
|
||||
|
||||
// =============================================================================
|
||||
|
||||
@@ -57,6 +57,7 @@ export interface NativeBindings {
|
||||
PhotonImage: NativePhotonImageConstructor;
|
||||
truncateToWidth(text: TextInput, maxWidth: number, ellipsis: TextInput, pad: boolean): string;
|
||||
sliceWithWidth(line: TextInput, startCol: number, length: number, strict: boolean): SliceWithWidthResult;
|
||||
visibleWidth(text: TextInput): number;
|
||||
extractSegments(
|
||||
line: TextInput,
|
||||
beforeEnd: number,
|
||||
|
||||
@@ -25,6 +25,13 @@ export function truncateToWidth(text: TextInput, maxWidth: number, ellipsis: Tex
|
||||
return native.truncateToWidth(text, maxWidth, ellipsis, pad);
|
||||
}
|
||||
|
||||
/**
|
||||
* Measure the visible width of text (excluding ANSI codes).
|
||||
*/
|
||||
export function visibleWidth(text: TextInput): number {
|
||||
return native.visibleWidth(text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Slice a range of visible columns from a line.
|
||||
*/
|
||||
|
||||
@@ -1,6 +1,13 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Changed
|
||||
|
||||
- Refactored `visibleWidth` function to use caching wrapper around new `visibleWidthRaw` implementation for improved performance
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed `extractAnsiCode` function from public API
|
||||
|
||||
## [9.6.1] - 2026-02-01
|
||||
### Changed
|
||||
|
||||
@@ -0,0 +1,187 @@
|
||||
/**
|
||||
* Benchmark: native visibleWidth vs Bun.stringWidth vs hybrid implementation
|
||||
*
|
||||
* Run: bun packages/tui/bench/visible-width.ts
|
||||
*/
|
||||
import { visibleWidth as nativeVisibleWidth } from "@oh-my-pi/pi-natives";
|
||||
import { visibleWidthRaw as hybridVisibleWidth } from "../src/utils";
|
||||
|
||||
const ITERATIONS = 10_000;
|
||||
const WARMUP = 500;
|
||||
|
||||
// Test cases covering different scenarios
|
||||
const samples = {
|
||||
// Pure ASCII - different lengths
|
||||
ascii_short: "hello",
|
||||
ascii_medium: "hello world this is a plain ASCII string with some words",
|
||||
ascii_long: "a".repeat(500),
|
||||
|
||||
// ANSI escape codes
|
||||
ansi_simple: "\x1b[31mred\x1b[0m",
|
||||
ansi_complex: "\x1b[31mred text\x1b[0m and \x1b[4munderlined content\x1b[24m with more \x1b[1;33;44mstyles\x1b[0m",
|
||||
ansi_nested: "\x1b[1m\x1b[31m\x1b[4mbold red underline\x1b[0m normal \x1b[32mgreen\x1b[0m",
|
||||
|
||||
// OSC 8 hyperlinks
|
||||
links: "prefix \x1b]8;;https://example.com\x07link text\x1b]8;;\x07 suffix",
|
||||
links_multiple:
|
||||
"Click \x1b]8;;https://a.com\x07here\x1b]8;;\x07 or \x1b]8;;https://b.com\x07there\x1b]8;;\x07 for info",
|
||||
|
||||
// Wide characters (CJK)
|
||||
cjk_short: "日本語",
|
||||
cjk_medium: "日本語のテキストとemoji",
|
||||
cjk_long: "日本語のテキストと中文字符和한국어문자混合在一起形成很长的字符串",
|
||||
|
||||
// Emoji
|
||||
emoji_simple: "👋🌍",
|
||||
emoji_complex: "Hello 👨👩👧👦 family! 🚀✨🎉 Let's go! 🇺🇸🏳️🌈",
|
||||
emoji_zwj: "👨💻👩🔬👨👩👧👦", // ZWJ sequences
|
||||
|
||||
// Mixed content
|
||||
mixed_short: "Hello 世界 🌍",
|
||||
mixed_medium: "\x1b[32mStatus:\x1b[0m 成功 ✓ (took 42ms)",
|
||||
mixed_long:
|
||||
"\x1b[1;34m[INFO]\x1b[0m Processing 日本語テキスト with emoji 🚀 and \x1b]8;;https://example.com\x07links\x1b]8;;\x07 完了",
|
||||
|
||||
// Edge cases
|
||||
tabs: "col1\tcol2\tcol3\tcol4",
|
||||
empty: "",
|
||||
newlines: "line1\nline2\nline3",
|
||||
control_chars: "text\x00with\x01control\x02chars",
|
||||
};
|
||||
|
||||
// Bun.stringWidth with ANSI stripping (what hybrid uses for short strings)
|
||||
function bunStringWidth(str: string): number {
|
||||
if (str.length === 0) return 0;
|
||||
|
||||
let clean = str;
|
||||
if (str.includes("\t")) {
|
||||
clean = clean.replace(/\t/g, " ");
|
||||
}
|
||||
if (clean.includes("\x1b")) {
|
||||
clean = clean.replace(/\x1b\[[0-9;]*[mGKHJ]/g, "");
|
||||
clean = clean.replace(/\x1b\]8;;[^\x07]*\x07/g, "");
|
||||
}
|
||||
return Bun.stringWidth(clean);
|
||||
}
|
||||
|
||||
interface BenchResult {
|
||||
name: string;
|
||||
totalMs: number;
|
||||
perOpUs: number;
|
||||
}
|
||||
|
||||
function bench(name: string, fn: () => void): BenchResult {
|
||||
// Warmup
|
||||
for (let i = 0; i < WARMUP; i++) fn();
|
||||
|
||||
const start = performance.now();
|
||||
for (let i = 0; i < ITERATIONS; i++) {
|
||||
fn();
|
||||
}
|
||||
const totalMs = performance.now() - start;
|
||||
const perOpUs = (totalMs / ITERATIONS) * 1000;
|
||||
|
||||
return { name, totalMs, perOpUs };
|
||||
}
|
||||
|
||||
function formatResult(r: BenchResult, baseline?: BenchResult): string {
|
||||
const perOp = r.perOpUs.toFixed(3);
|
||||
if (baseline && baseline !== r) {
|
||||
const ratio = r.perOpUs / baseline.perOpUs;
|
||||
const indicator = ratio < 1 ? "faster" : "slower";
|
||||
return `${r.name.padEnd(20)} ${r.totalMs.toFixed(2).padStart(8)}ms ${perOp.padStart(8)}µs/op ${ratio.toFixed(2)}x ${indicator}`;
|
||||
}
|
||||
return `${r.name.padEnd(20)} ${r.totalMs.toFixed(2).padStart(8)}ms ${perOp.padStart(8)}µs/op (baseline)`;
|
||||
}
|
||||
|
||||
console.log(`\n${"=".repeat(80)}`);
|
||||
console.log(`visibleWidth benchmark: ${ITERATIONS.toLocaleString()} iterations, ${WARMUP} warmup`);
|
||||
console.log(`${"=".repeat(80)}\n`);
|
||||
|
||||
for (const [sampleName, sample] of Object.entries(samples)) {
|
||||
console.log(`\n--- ${sampleName} (len=${sample.length}) ---`);
|
||||
if (sample.length > 0 && sample.length < 80) {
|
||||
// Show sample for short strings (escape non-printable)
|
||||
const display = sample.replace(/\x1b/g, "\\e").replace(/\x07/g, "\\a");
|
||||
console.log(` "${display}"`);
|
||||
}
|
||||
|
||||
const results: BenchResult[] = [];
|
||||
|
||||
results.push(
|
||||
bench("native", () => {
|
||||
nativeVisibleWidth(sample);
|
||||
}),
|
||||
);
|
||||
|
||||
results.push(
|
||||
bench("bun+strip", () => {
|
||||
bunStringWidth(sample);
|
||||
}),
|
||||
);
|
||||
|
||||
results.push(
|
||||
bench("hybrid", () => {
|
||||
hybridVisibleWidth(sample);
|
||||
}),
|
||||
);
|
||||
|
||||
// Find fastest as baseline
|
||||
const baseline = results.reduce((a, b) => (a.perOpUs < b.perOpUs ? a : b));
|
||||
|
||||
console.log();
|
||||
for (const r of results) {
|
||||
console.log(` ${formatResult(r, baseline)}`);
|
||||
}
|
||||
|
||||
// Verify correctness
|
||||
const nativeResult = nativeVisibleWidth(sample);
|
||||
const bunResult = bunStringWidth(sample);
|
||||
const hybridResult = hybridVisibleWidth(sample);
|
||||
|
||||
if (nativeResult !== hybridResult) {
|
||||
console.log(` ⚠️ MISMATCH: native=${nativeResult}, hybrid=${hybridResult}`);
|
||||
}
|
||||
if (bunResult !== hybridResult && !sample.includes("\x00")) {
|
||||
// Control chars can differ
|
||||
console.log(` ⚠️ MISMATCH: bun=${bunResult}, hybrid=${hybridResult}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`\n${"=".repeat(80)}`);
|
||||
console.log("Summary");
|
||||
console.log(`${"=".repeat(80)}\n`);
|
||||
|
||||
// Aggregate by category
|
||||
const categories = {
|
||||
ascii: ["ascii_short", "ascii_medium", "ascii_long"],
|
||||
ansi: ["ansi_simple", "ansi_complex", "ansi_nested"],
|
||||
links: ["links", "links_multiple"],
|
||||
cjk: ["cjk_short", "cjk_medium", "cjk_long"],
|
||||
emoji: ["emoji_simple", "emoji_complex", "emoji_zwj"],
|
||||
mixed: ["mixed_short", "mixed_medium", "mixed_long"],
|
||||
};
|
||||
|
||||
for (const [category, sampleNames] of Object.entries(categories)) {
|
||||
const categoryResults = { native: 0, bun: 0, hybrid: 0 };
|
||||
|
||||
for (const name of sampleNames) {
|
||||
const sample = samples[name as keyof typeof samples];
|
||||
|
||||
const nativeTime = bench("", () => nativeVisibleWidth(sample)).perOpUs;
|
||||
const bunTime = bench("", () => bunStringWidth(sample)).perOpUs;
|
||||
const hybridTime = bench("", () => hybridVisibleWidth(sample)).perOpUs;
|
||||
|
||||
categoryResults.native += nativeTime;
|
||||
categoryResults.bun += bunTime;
|
||||
categoryResults.hybrid += hybridTime;
|
||||
}
|
||||
|
||||
const fastest = Math.min(categoryResults.native, categoryResults.bun, categoryResults.hybrid);
|
||||
const winner =
|
||||
fastest === categoryResults.native ? "native" : fastest === categoryResults.bun ? "bun+strip" : "hybrid";
|
||||
|
||||
console.log(
|
||||
`${category.padEnd(10)} native: ${categoryResults.native.toFixed(1)}µs bun: ${categoryResults.bun.toFixed(1)}µs hybrid: ${categoryResults.hybrid.toFixed(1)}µs → ${winner} wins`,
|
||||
);
|
||||
}
|
||||
+12
-36
@@ -33,7 +33,7 @@ const widthCache = new Map<string, number>();
|
||||
/**
|
||||
* Calculate the visible width of a string in terminal columns.
|
||||
*/
|
||||
export function visibleWidth(str: string): number {
|
||||
export function visibleWidthRaw(str: string): number {
|
||||
if (str.length === 0) {
|
||||
return 0;
|
||||
}
|
||||
@@ -52,6 +52,16 @@ export function visibleWidth(str: string): number {
|
||||
if (isPureAscii) {
|
||||
return str.length + tabLength;
|
||||
}
|
||||
return Bun.stringWidth(str) + tabLength;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate the visible width of a string in terminal columns.
|
||||
*/
|
||||
export function visibleWidth(str: string): number {
|
||||
if (str.length === 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Check cache
|
||||
const cached = widthCache.get(str);
|
||||
@@ -59,8 +69,7 @@ export function visibleWidth(str: string): number {
|
||||
return cached;
|
||||
}
|
||||
|
||||
// Cache result
|
||||
const width = Bun.stringWidth(str) + tabLength;
|
||||
const width = visibleWidthRaw(str);
|
||||
if (widthCache.size >= WIDTH_CACHE_SIZE) {
|
||||
const firstKey = widthCache.keys().next().value;
|
||||
if (firstKey !== undefined) {
|
||||
@@ -72,39 +81,6 @@ export function visibleWidth(str: string): number {
|
||||
return width;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract ANSI escape sequences from a string at the given position.
|
||||
*/
|
||||
export function extractAnsiCode(str: string, pos: number): { code: string; length: number } | null {
|
||||
if (pos >= str.length || str[pos] !== "\x1b") return null;
|
||||
|
||||
const next = str[pos + 1];
|
||||
|
||||
// CSI sequence: ESC [ ... m/G/K/H/J
|
||||
if (next === "[") {
|
||||
let j = pos + 2;
|
||||
while (j < str.length && !/[mGKHJ]/.test(str[j]!)) j++;
|
||||
if (j < str.length) return { code: str.substring(pos, j + 1), length: j + 1 - pos };
|
||||
return null;
|
||||
}
|
||||
|
||||
// OSC sequence: ESC ] ... BEL or ESC ] ... ST (ESC \)
|
||||
// Used for hyperlinks (OSC 8), window titles, etc.
|
||||
if (next === "]") {
|
||||
let j = pos + 2;
|
||||
while (j < str.length) {
|
||||
if (str[j] === "\x07") return { code: str.substring(pos, j + 1), length: j + 1 - pos };
|
||||
if (str[j] === "\x1b" && str[j + 1] === "\\") {
|
||||
return { code: str.substring(pos, j + 2), length: j + 2 - pos };
|
||||
}
|
||||
j++;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
const WRAP_OPTIONS = { wordWrap: true, hard: true, trim: false } as const;
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user