d9a3291641
- Replaced external JSON/JSONL parsing libraries with Bun's built-in JSON5 and JSONL APIs across all packages. - Removed dependencies: json5, ndjson, get-east-asian-width, json-stringify-safe, split2, through2, and @types/ndjson. - Replaced custom text width and ANSI wrapping implementations with Bun.stringWidth() and Bun.wrapAnsi() APIs. - Updated Bun type definitions from ^1.2.17 to ^1.2.18 and bun-types from ^1.3.5 to ^1.3.7. - Refactored JSONL parsing throughout codebase to use Bun.JSONL.parse() and Bun.JSONL.parseChunk() with improved buffer management and error handling. - Updated test assertions to use Bun.stringWidth() for dynamic width calculations instead of hardcoded values.
585 lines
15 KiB
TypeScript
585 lines
15 KiB
TypeScript
// Grapheme segmenter (shared instance)
|
|
const segmenter = new Intl.Segmenter(undefined, { granularity: "grapheme" });
|
|
|
|
/**
|
|
* Get the shared grapheme segmenter instance.
|
|
*/
|
|
export function getSegmenter(): Intl.Segmenter {
|
|
return segmenter;
|
|
}
|
|
|
|
|
|
// Cache for non-ASCII strings
|
|
const WIDTH_CACHE_SIZE = 512;
|
|
const widthCache = new Map<string, number>();
|
|
|
|
|
|
/**
|
|
* Calculate the visible width of a string in terminal columns.
|
|
*/
|
|
export function visibleWidth(str: string): number {
|
|
if (str.length === 0) {
|
|
return 0;
|
|
}
|
|
|
|
// Fast path: pure ASCII printable
|
|
let isPureAscii = true;
|
|
for (let i = 0; i < str.length; i++) {
|
|
const code = str.charCodeAt(i);
|
|
if (code < 0x20 || code > 0x7e) {
|
|
isPureAscii = false;
|
|
break;
|
|
}
|
|
}
|
|
if (isPureAscii) {
|
|
return str.length;
|
|
}
|
|
|
|
// Check cache
|
|
const cached = widthCache.get(str);
|
|
if (cached !== undefined) {
|
|
return cached;
|
|
}
|
|
|
|
// Normalize: tabs to 3 spaces, strip ANSI escape codes
|
|
let clean = str;
|
|
if (str.includes("\t")) {
|
|
clean = clean.replace(/\t/g, " ");
|
|
}
|
|
if (clean.includes("\x1b")) {
|
|
// Strip SGR codes (\x1b[...m) and cursor codes (\x1b[...G/K/H/J)
|
|
clean = clean.replace(/\x1b\[[0-9;]*[mGKHJ]/g, "");
|
|
// Strip OSC 8 hyperlinks: \x1b]8;;URL\x07 and \x1b]8;;\x07
|
|
clean = clean.replace(/\x1b\]8;;[^\x07]*\x07/g, "");
|
|
}
|
|
|
|
|
|
const width = Bun.stringWidth(clean);
|
|
|
|
// Cache result
|
|
if (widthCache.size >= WIDTH_CACHE_SIZE) {
|
|
const firstKey = widthCache.keys().next().value;
|
|
if (firstKey !== undefined) {
|
|
widthCache.delete(firstKey);
|
|
}
|
|
}
|
|
widthCache.set(str, width);
|
|
|
|
return width;
|
|
}
|
|
|
|
/**
|
|
* Extract ANSI escape sequences from a string at the given position.
|
|
*/
|
|
export function extractAnsiCode(str: string, pos: number): { code: string; length: number } | null {
|
|
if (pos >= str.length || str[pos] !== "\x1b") return null;
|
|
|
|
const next = str[pos + 1];
|
|
|
|
// CSI sequence: ESC [ ... m/G/K/H/J
|
|
if (next === "[") {
|
|
let j = pos + 2;
|
|
while (j < str.length && !/[mGKHJ]/.test(str[j]!)) j++;
|
|
if (j < str.length) return { code: str.substring(pos, j + 1), length: j + 1 - pos };
|
|
return null;
|
|
}
|
|
|
|
// OSC sequence: ESC ] ... BEL or ESC ] ... ST (ESC \)
|
|
// Used for hyperlinks (OSC 8), window titles, etc.
|
|
if (next === "]") {
|
|
let j = pos + 2;
|
|
while (j < str.length) {
|
|
if (str[j] === "\x07") return { code: str.substring(pos, j + 1), length: j + 1 - pos };
|
|
if (str[j] === "\x1b" && str[j + 1] === "\\") {
|
|
return { code: str.substring(pos, j + 2), length: j + 2 - pos };
|
|
}
|
|
j++;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
/**
|
|
* Track active ANSI SGR codes to preserve styling across line breaks.
|
|
*/
|
|
class AnsiCodeTracker {
|
|
// Track individual attributes separately so we can reset them specifically
|
|
private bold = false;
|
|
private dim = false;
|
|
private italic = false;
|
|
private underline = false;
|
|
private blink = false;
|
|
private inverse = false;
|
|
private hidden = false;
|
|
private strikethrough = false;
|
|
private fgColor: string | null = null; // Stores the full code like "31" or "38;5;240"
|
|
private bgColor: string | null = null; // Stores the full code like "41" or "48;5;240"
|
|
|
|
process(ansiCode: string): void {
|
|
if (!ansiCode.endsWith("m")) {
|
|
return;
|
|
}
|
|
|
|
// Extract the parameters between \x1b[ and m
|
|
const match = ansiCode.match(/\x1b\[([\d;]*)m/);
|
|
if (!match) return;
|
|
|
|
const params = match[1];
|
|
if (params === "" || params === "0") {
|
|
// Full reset
|
|
this.reset();
|
|
return;
|
|
}
|
|
|
|
// Parse parameters (can be semicolon-separated)
|
|
const parts = params.split(";");
|
|
let i = 0;
|
|
while (i < parts.length) {
|
|
const code = Number.parseInt(parts[i], 10);
|
|
|
|
// Handle 256-color and RGB codes which consume multiple parameters
|
|
if (code === 38 || code === 48) {
|
|
// 38;5;N (256 color fg) or 38;2;R;G;B (RGB fg)
|
|
// 48;5;N (256 color bg) or 48;2;R;G;B (RGB bg)
|
|
if (parts[i + 1] === "5" && parts[i + 2] !== undefined) {
|
|
// 256 color: 38;5;N or 48;5;N
|
|
const colorCode = `${parts[i]};${parts[i + 1]};${parts[i + 2]}`;
|
|
if (code === 38) {
|
|
this.fgColor = colorCode;
|
|
} else {
|
|
this.bgColor = colorCode;
|
|
}
|
|
i += 3;
|
|
continue;
|
|
} else if (parts[i + 1] === "2" && parts[i + 4] !== undefined) {
|
|
// RGB color: 38;2;R;G;B or 48;2;R;G;B
|
|
const colorCode = `${parts[i]};${parts[i + 1]};${parts[i + 2]};${parts[i + 3]};${parts[i + 4]}`;
|
|
if (code === 38) {
|
|
this.fgColor = colorCode;
|
|
} else {
|
|
this.bgColor = colorCode;
|
|
}
|
|
i += 5;
|
|
continue;
|
|
}
|
|
}
|
|
|
|
// Standard SGR codes
|
|
switch (code) {
|
|
case 0:
|
|
this.reset();
|
|
break;
|
|
case 1:
|
|
this.bold = true;
|
|
break;
|
|
case 2:
|
|
this.dim = true;
|
|
break;
|
|
case 3:
|
|
this.italic = true;
|
|
break;
|
|
case 4:
|
|
this.underline = true;
|
|
break;
|
|
case 5:
|
|
this.blink = true;
|
|
break;
|
|
case 7:
|
|
this.inverse = true;
|
|
break;
|
|
case 8:
|
|
this.hidden = true;
|
|
break;
|
|
case 9:
|
|
this.strikethrough = true;
|
|
break;
|
|
case 21:
|
|
this.bold = false;
|
|
break; // Some terminals
|
|
case 22:
|
|
this.bold = false;
|
|
this.dim = false;
|
|
break;
|
|
case 23:
|
|
this.italic = false;
|
|
break;
|
|
case 24:
|
|
this.underline = false;
|
|
break;
|
|
case 25:
|
|
this.blink = false;
|
|
break;
|
|
case 27:
|
|
this.inverse = false;
|
|
break;
|
|
case 28:
|
|
this.hidden = false;
|
|
break;
|
|
case 29:
|
|
this.strikethrough = false;
|
|
break;
|
|
case 39:
|
|
this.fgColor = null;
|
|
break; // Default fg
|
|
case 49:
|
|
this.bgColor = null;
|
|
break; // Default bg
|
|
default:
|
|
// Standard foreground colors 30-37, 90-97
|
|
if ((code >= 30 && code <= 37) || (code >= 90 && code <= 97)) {
|
|
this.fgColor = String(code);
|
|
}
|
|
// Standard background colors 40-47, 100-107
|
|
else if ((code >= 40 && code <= 47) || (code >= 100 && code <= 107)) {
|
|
this.bgColor = String(code);
|
|
}
|
|
break;
|
|
}
|
|
i++;
|
|
}
|
|
}
|
|
|
|
private reset(): void {
|
|
this.bold = false;
|
|
this.dim = false;
|
|
this.italic = false;
|
|
this.underline = false;
|
|
this.blink = false;
|
|
this.inverse = false;
|
|
this.hidden = false;
|
|
this.strikethrough = false;
|
|
this.fgColor = null;
|
|
this.bgColor = null;
|
|
}
|
|
|
|
/** Clear all state for reuse. */
|
|
clear(): void {
|
|
this.reset();
|
|
}
|
|
|
|
getActiveCodes(): string {
|
|
const codes: string[] = [];
|
|
if (this.bold) codes.push("1");
|
|
if (this.dim) codes.push("2");
|
|
if (this.italic) codes.push("3");
|
|
if (this.underline) codes.push("4");
|
|
if (this.blink) codes.push("5");
|
|
if (this.inverse) codes.push("7");
|
|
if (this.hidden) codes.push("8");
|
|
if (this.strikethrough) codes.push("9");
|
|
if (this.fgColor) codes.push(this.fgColor);
|
|
if (this.bgColor) codes.push(this.bgColor);
|
|
|
|
if (codes.length === 0) return "";
|
|
return `\x1b[${codes.join(";")}m`;
|
|
}
|
|
|
|
hasActiveCodes(): boolean {
|
|
return (
|
|
this.bold ||
|
|
this.dim ||
|
|
this.italic ||
|
|
this.underline ||
|
|
this.blink ||
|
|
this.inverse ||
|
|
this.hidden ||
|
|
this.strikethrough ||
|
|
this.fgColor !== null ||
|
|
this.bgColor !== null
|
|
);
|
|
}
|
|
|
|
/**
|
|
* Get reset codes for attributes that need to be turned off at line end,
|
|
* specifically underline which bleeds into padding.
|
|
* Returns empty string if no problematic attributes are active.
|
|
*/
|
|
getLineEndReset(): string {
|
|
// Only underline causes visual bleeding into padding
|
|
// Other attributes like colors don't visually bleed to padding
|
|
if (this.underline) {
|
|
return "\x1b[24m"; // Underline off only
|
|
}
|
|
return "";
|
|
}
|
|
}
|
|
|
|
|
|
|
|
/**
|
|
* Wrap text with ANSI codes preserved.
|
|
*
|
|
* ONLY does word wrapping - NO padding, NO background colors.
|
|
* Returns lines where each line is <= width visible chars.
|
|
* Active ANSI codes are preserved across line breaks.
|
|
*
|
|
* @param text - Text to wrap (may contain ANSI codes and newlines)
|
|
* @param width - Maximum visible width per line
|
|
* @returns Array of wrapped lines (NOT padded to width)
|
|
*/
|
|
export function wrapTextWithAnsi(text: string, width: number): string[] {
|
|
if (!text) {
|
|
return [""];
|
|
}
|
|
|
|
return Bun.wrapAnsi(text, width, { wordWrap: true, hard: true, trim: false }).split("\n");
|
|
}
|
|
|
|
const PUNCTUATION_REGEX = /[(){}[\]<>.,;:'"!?+\-=*/\\|&%^$#@~`]/;
|
|
|
|
/**
|
|
* Check if a character is whitespace.
|
|
*/
|
|
export function isWhitespaceChar(char: string): boolean {
|
|
return /\s/.test(char);
|
|
}
|
|
|
|
/**
|
|
* Check if a character is punctuation.
|
|
*/
|
|
export function isPunctuationChar(char: string): boolean {
|
|
return PUNCTUATION_REGEX.test(char);
|
|
}
|
|
|
|
/**
|
|
* Apply background color to a line, padding to full width.
|
|
*
|
|
* @param line - Line of text (may contain ANSI codes)
|
|
* @param width - Total width to pad to
|
|
* @param bgFn - Background color function
|
|
* @returns Line with background applied and padded to width
|
|
*/
|
|
export function applyBackgroundToLine(line: string, width: number, bgFn: (text: string) => string): string {
|
|
// Calculate padding needed
|
|
const visibleLen = visibleWidth(line);
|
|
const paddingNeeded = Math.max(0, width - visibleLen);
|
|
const padding = " ".repeat(paddingNeeded);
|
|
|
|
// Apply background to content + padding
|
|
const withPadding = line + padding;
|
|
return bgFn(withPadding);
|
|
}
|
|
|
|
/**
|
|
* Truncate text to fit within a maximum visible width, adding ellipsis if needed.
|
|
* Optionally pad with spaces to reach exactly maxWidth.
|
|
* Properly handles ANSI escape codes (they don't count toward width).
|
|
*
|
|
* @param text - Text to truncate (may contain ANSI codes)
|
|
* @param maxWidth - Maximum visible width
|
|
* @param ellipsis - Ellipsis string to append when truncating (default: "…")
|
|
* @param pad - If true, pad result with spaces to exactly maxWidth (default: false)
|
|
* @returns Truncated text, optionally padded to exactly maxWidth
|
|
*/
|
|
export function truncateToWidth(text: string, maxWidth: number, ellipsis: string = "…", pad: boolean = false): string {
|
|
const textVisibleWidth = visibleWidth(text);
|
|
|
|
if (textVisibleWidth <= maxWidth) {
|
|
return pad ? text + " ".repeat(maxWidth - textVisibleWidth) : text;
|
|
}
|
|
|
|
const ellipsisWidth = visibleWidth(ellipsis);
|
|
const targetWidth = maxWidth - ellipsisWidth;
|
|
|
|
if (targetWidth <= 0) {
|
|
return ellipsis.substring(0, maxWidth);
|
|
}
|
|
|
|
// Separate ANSI codes from visible content using grapheme segmentation
|
|
let i = 0;
|
|
const segments: Array<{ type: "ansi" | "grapheme"; value: string }> = [];
|
|
|
|
while (i < text.length) {
|
|
const ansiResult = extractAnsiCode(text, i);
|
|
if (ansiResult) {
|
|
segments.push({ type: "ansi", value: ansiResult.code });
|
|
i += ansiResult.length;
|
|
} else {
|
|
// Find the next ANSI code or end of string
|
|
let end = i;
|
|
while (end < text.length) {
|
|
const nextAnsi = extractAnsiCode(text, end);
|
|
if (nextAnsi) break;
|
|
end++;
|
|
}
|
|
// Segment this non-ANSI portion into graphemes
|
|
const textPortion = text.slice(i, end);
|
|
for (const seg of segmenter.segment(textPortion)) {
|
|
segments.push({ type: "grapheme", value: seg.segment });
|
|
}
|
|
i = end;
|
|
}
|
|
}
|
|
|
|
// Build truncated string from segments
|
|
let result = "";
|
|
let currentWidth = 0;
|
|
|
|
for (const seg of segments) {
|
|
if (seg.type === "ansi") {
|
|
result += seg.value;
|
|
continue;
|
|
}
|
|
|
|
const grapheme = seg.value;
|
|
// Skip empty graphemes to avoid issues with string-width calculation
|
|
if (!grapheme) continue;
|
|
|
|
const graphemeWidth = visibleWidth(grapheme);
|
|
|
|
if (currentWidth + graphemeWidth > targetWidth) {
|
|
break;
|
|
}
|
|
|
|
result += grapheme;
|
|
currentWidth += graphemeWidth;
|
|
}
|
|
|
|
// Add reset code before ellipsis to prevent styling leaking into it
|
|
const truncated = `${result}\x1b[0m${ellipsis}`;
|
|
if (pad) {
|
|
const truncatedWidth = visibleWidth(truncated);
|
|
return truncated + " ".repeat(Math.max(0, maxWidth - truncatedWidth));
|
|
}
|
|
return truncated;
|
|
}
|
|
|
|
/**
|
|
* Extract a range of visible columns from a line. Handles ANSI codes and wide chars.
|
|
* @param strict - If true, exclude wide chars at boundary that would extend past the range
|
|
*/
|
|
export function sliceByColumn(line: string, startCol: number, length: number, strict = false): string {
|
|
return sliceWithWidth(line, startCol, length, strict).text;
|
|
}
|
|
|
|
/** Like sliceByColumn but also returns the actual visible width of the result. */
|
|
export function sliceWithWidth(
|
|
line: string,
|
|
startCol: number,
|
|
length: number,
|
|
strict = false,
|
|
): { text: string; width: number } {
|
|
if (length <= 0) return { text: "", width: 0 };
|
|
const endCol = startCol + length;
|
|
let result = "",
|
|
resultWidth = 0,
|
|
currentCol = 0,
|
|
i = 0,
|
|
pendingAnsi = "";
|
|
|
|
while (i < line.length) {
|
|
const ansi = extractAnsiCode(line, i);
|
|
if (ansi) {
|
|
if (currentCol >= startCol && currentCol < endCol) result += ansi.code;
|
|
else if (currentCol < startCol) pendingAnsi += ansi.code;
|
|
i += ansi.length;
|
|
continue;
|
|
}
|
|
|
|
let textEnd = i;
|
|
while (textEnd < line.length && !extractAnsiCode(line, textEnd)) textEnd++;
|
|
|
|
for (const { segment } of segmenter.segment(line.slice(i, textEnd))) {
|
|
const w = visibleWidth(segment);
|
|
const inRange = currentCol >= startCol && currentCol < endCol;
|
|
const fits = !strict || currentCol + w <= endCol;
|
|
if (inRange && fits) {
|
|
if (pendingAnsi) {
|
|
result += pendingAnsi;
|
|
pendingAnsi = "";
|
|
}
|
|
result += segment;
|
|
resultWidth += w;
|
|
}
|
|
currentCol += w;
|
|
if (currentCol >= endCol) break;
|
|
}
|
|
i = textEnd;
|
|
if (currentCol >= endCol) break;
|
|
}
|
|
return { text: result, width: resultWidth };
|
|
}
|
|
|
|
// Pooled tracker instance for extractSegments (avoids allocation per call)
|
|
const pooledStyleTracker = new AnsiCodeTracker();
|
|
|
|
/**
|
|
* Extract "before" and "after" segments from a line in a single pass.
|
|
* Used for overlay compositing where we need content before and after the overlay region.
|
|
* Preserves styling from before the overlay that should affect content after it.
|
|
*/
|
|
export function extractSegments(
|
|
line: string,
|
|
beforeEnd: number,
|
|
afterStart: number,
|
|
afterLen: number,
|
|
strictAfter = false,
|
|
): { before: string; beforeWidth: number; after: string; afterWidth: number } {
|
|
let before = "",
|
|
beforeWidth = 0,
|
|
after = "",
|
|
afterWidth = 0;
|
|
let currentCol = 0,
|
|
i = 0;
|
|
let pendingAnsiBefore = "";
|
|
let afterStarted = false;
|
|
const afterEnd = afterStart + afterLen;
|
|
|
|
// Track styling state so "after" inherits styling from before the overlay
|
|
pooledStyleTracker.clear();
|
|
|
|
while (i < line.length) {
|
|
const ansi = extractAnsiCode(line, i);
|
|
if (ansi) {
|
|
// Track all SGR codes to know styling state at afterStart
|
|
pooledStyleTracker.process(ansi.code);
|
|
// Include ANSI codes in their respective segments
|
|
if (currentCol < beforeEnd) {
|
|
pendingAnsiBefore += ansi.code;
|
|
} else if (currentCol >= afterStart && currentCol < afterEnd && afterStarted) {
|
|
// Only include after we've started "after" (styling already prepended)
|
|
after += ansi.code;
|
|
}
|
|
i += ansi.length;
|
|
continue;
|
|
}
|
|
|
|
let textEnd = i;
|
|
while (textEnd < line.length && !extractAnsiCode(line, textEnd)) textEnd++;
|
|
|
|
for (const { segment } of segmenter.segment(line.slice(i, textEnd))) {
|
|
const w = visibleWidth(segment);
|
|
|
|
if (currentCol < beforeEnd) {
|
|
if (pendingAnsiBefore) {
|
|
before += pendingAnsiBefore;
|
|
pendingAnsiBefore = "";
|
|
}
|
|
before += segment;
|
|
beforeWidth += w;
|
|
} else if (currentCol >= afterStart && currentCol < afterEnd) {
|
|
const fits = !strictAfter || currentCol + w <= afterEnd;
|
|
if (fits) {
|
|
// On first "after" grapheme, prepend inherited styling from before overlay
|
|
if (!afterStarted) {
|
|
after += pooledStyleTracker.getActiveCodes();
|
|
afterStarted = true;
|
|
}
|
|
after += segment;
|
|
afterWidth += w;
|
|
}
|
|
}
|
|
|
|
currentCol += w;
|
|
// Early exit: done with "before" only, or done with both segments
|
|
if (afterLen <= 0 ? currentCol >= beforeEnd : currentCol >= afterEnd) break;
|
|
}
|
|
i = textEnd;
|
|
if (afterLen <= 0 ? currentCol >= beforeEnd : currentCol >= afterEnd) break;
|
|
}
|
|
|
|
return { before, beforeWidth, after, afterWidth };
|
|
}
|