diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index bfde4e296..29b62bdc9 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,16 @@ ## [Unreleased] +### Added + +- Added bulk-input fast path and iterative processing for bracketed paste in the editor +- Added windowed incremental lexing for large markdown documents + +### Changed + +- Optimized markdown URL tokenizer gate, inline math start scan, and autolink scheme scan for performance +- Deduplicated terminal cursor-visibility writes to skip redundant escape sequences + ## [17.1.6] - 2026-07-27 ### Fixed diff --git a/packages/tui/README.md b/packages/tui/README.md index e32729d48..950b0cc49 100644 --- a/packages/tui/README.md +++ b/packages/tui/README.md @@ -528,8 +528,8 @@ interface Terminal { get columns(): number; get rows(): number; moveBy(lines: number): void; - hideCursor(): void; - showCursor(): void; + hideCursor(force?: boolean): void; + showCursor(force?: boolean): void; clearLine(): void; clearFromCursor(): void; clearScreen(): void; diff --git a/packages/tui/bench/markdown-stream.ts b/packages/tui/bench/markdown-stream.ts new file mode 100644 index 000000000..69565e1f6 --- /dev/null +++ b/packages/tui/bench/markdown-stream.ts @@ -0,0 +1,92 @@ +/** + * Streaming markdown render benchmark. + * + * Simulates a model streaming a long markdown message into one reused + * `Markdown` component (the interactive-mode hot path): the text grows in + * fixed-size deltas and the component re-renders after each delta. + * + * Exercises the profile hotspots from the 2026-07 capture: marked's GFM `url` + * tokenizer (73.3% self), inline extension `start()` scans, emStrong, and the + * streaming stable-prefix freeze (`#freezeStablePrefix`). + * + * Run: bun packages/tui/bench/markdown-stream.ts + */ +import { clearRenderCache, Markdown } from "../src/components/markdown"; +import { defaultMarkdownTheme } from "../test/test-themes"; + +const WIDTH = 100; +const DELTA = 64; // chars revealed per streaming step + +// --- Fixtures ------------------------------------------------------------- + +/** Long bullet list — the shape that defeats prefix freezing (list guard). */ +function bulletList(items: number): string { + const lines: string[] = []; + for (let i = 0; i < items; i++) { + lines.push( + `- \`packages/tui/src/components/markdown_component_${i}.ts\` handles the ` + + `stable_prefix_freeze_path_${i} and re-lexes only the unfrozen tail, see ` + + `https://github.com/can1357/oh-my-pi/issues/${1000 + i} for details on token_${i}.`, + ); + } + return `${lines.join("\n")}\n\n`; +} + +/** Prose dense in email-branch pathology: long `[A-Za-z0-9._+-]+` runs with no `@`. */ +function identifierProse(paragraphs: number): string { + const parts: string[] = []; + for (let i = 0; i < paragraphs; i++) { + parts.push( + `The resolver maps session_listing.scan_session_file.header_cache_v${i} onto ` + + `auth_broker.remote_store.filter_usage_reports_${i} while user${i}@example.com and ` + + `www.example${i}.org stay autolinked; identifiers like RENDER_CACHE_MAX_ENTRY_SIZE_${i} ` + + `and freeze.stable.prefix.tokens.v${i} must parse as plain *text* with **no** backtracking.`, + ); + } + return `${parts.join("\n\n")}\n\n`; +} + +function fences(count: number): string { + const parts: string[] = []; + for (let i = 0; i < count; i++) { + parts.push("```ts\nconst x_" + i + " = await fetch(\"https://api.example.com/v1/usage\");\n```\n"); + } + return `${parts.join("\n")}\n`; +} + +const DOC = identifierProse(20) + bulletList(120) + fences(8) + identifierProse(20) + bulletList(80); + +// --- Bench ---------------------------------------------------------------- + +function streamOnce(text: string): number { + clearRenderCache(); + const component = new Markdown("", 0, 0, defaultMarkdownTheme); + component.transientRenderCache = true; + const start = Bun.nanoseconds(); + for (let len = DELTA; len < text.length; len += DELTA) { + component.setText(text.slice(0, len)); + component.render(WIDTH); + } + component.setText(text); + component.render(WIDTH); + return (Bun.nanoseconds() - start) / 1e6; +} + +function coldOnce(text: string): number { + clearRenderCache(); + const start = Bun.nanoseconds(); + new Markdown(text, 0, 0, defaultMarkdownTheme).render(WIDTH); + return (Bun.nanoseconds() - start) / 1e6; +} + +console.log(`doc: ${DOC.length} chars, ${Math.ceil(DOC.length / DELTA)} streaming steps, width ${WIDTH}`); +// Warmup (JIT + regex compilation) +streamOnce(DOC.slice(0, 4096)); + +const cold = coldOnce(DOC); +console.log(`cold full render: ${cold.toFixed(1)}ms`); + +const runs: number[] = []; +for (let i = 0; i < 3; i++) runs.push(streamOnce(DOC)); +runs.sort((a, b) => a - b); +console.log(`streamed render (${runs.length} runs): min ${runs[0]!.toFixed(1)}ms, median ${runs[1]!.toFixed(1)}ms`); diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 1839d03bc..3e43a14d0 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -6,8 +6,8 @@ import { midPromptSkillTokenMatches, } from "../autocomplete"; import { BracketedPasteHandler, decodeReencodedPasteControls } from "../bracketed-paste"; -import { getKeybindings, type KeybindingsManager } from "../keybindings"; -import { extractPrintableText, matchesKey } from "../keys"; +import { canonicalKeyId, getKeybindings, type KeybindingsManager } from "../keybindings"; +import { extractPrintableText, matchesKey, parseKey } from "../keys"; import { KillRing } from "../kill-ring"; import type { SymbolTheme } from "../symbols"; import { type Component, CURSOR_MARKER, type Focusable } from "../tui"; @@ -341,6 +341,17 @@ function maxSegmentVisualCol(text: string, isLastSegment: boolean): number { return isLastSegment ? total : Math.max(0, total - lastWidth); } +/** True when every code unit is plain printable text: no C0 controls (so no + * ESC/CR/LF/TAB), no DEL, no C1 range — the same set `extractPrintableText` + * rejects. Such a run can never encode a key sequence. */ +function isPlainTextRun(data: string): boolean { + for (let i = 0; i < data.length; i++) { + const code = data.charCodeAt(i); + if (code < 0x20 || code === 0x7f || (code >= 0x80 && code <= 0x9f)) return false; + } + return true; +} + const DEFAULT_PAGE_SCROLL_LINES = 10; const MAX_UNDO_STACK = 100; @@ -1126,12 +1137,30 @@ export class Editor implements Component, Focusable { } handleInput(data: string): void { + // Iterative, not recursive: the bytes trailing a completed bracketed + // paste (which may themselves contain further pastes) loop back here, + // so a fragmented paste stream can never grow the call stack. + let next: string | undefined = data; + while (next !== undefined && next.length > 0) { + next = this.#handleInputChunk(next); + } + } + + /** Process one input chunk. Returns the unconsumed tail of a completed paste, if any. */ + #handleInputChunk(data: string): string | undefined { const kb = getKeybindings(); + // Parse the sequence once; every binding probe below is then a set + // lookup instead of re-parsing `data` per probe (~35 probes per key). + const parsedKey = parseKey(data); + const canonical = parsedKey === undefined ? undefined : canonicalKeyId(parsedKey); // Handle character jump mode (awaiting next character to jump to) if (this.#jumpMode !== null) { // Cancel if the hotkey is pressed again - if (kb.matches(data, "tui.editor.jumpForward") || kb.matches(data, "tui.editor.jumpBackward")) { + if ( + kb.matchesCanonical(canonical, "tui.editor.jumpForward") || + kb.matchesCanonical(canonical, "tui.editor.jumpBackward") + ) { this.#jumpMode = null; return; } @@ -1154,12 +1183,23 @@ export class Editor implements Component, Focusable { if (paste.pasteContent !== undefined) { this.#handlePaste(paste.pasteContent); if (paste.remaining.length > 0) { - this.handleInput(paste.remaining); + return paste.remaining; } } return; } + // Bulk printable fast path: a multi-scalar run of plain text (paste + // remainder, batched stdin) parses to no key, so no binding probe or + // special-key branch below can consume it — it always falls through to + // one #insertCharacter call. Take that path directly and skip the + // dispatch cascade. Runs containing ESC or control bytes (including + // \r/\n) keep the full path: those bytes carry key semantics. + if (canonical === undefined && data.length > 1 && isPlainTextRun(data)) { + this.#insertCharacter(data); + return; + } + // Handle special key combinations first // Ctrl+C is reserved by parent components for app-level handling. @@ -1170,7 +1210,7 @@ export class Editor implements Component, Focusable { } // Undo - if (kb.matches(data, "tui.editor.undo")) { + if (kb.matchesCanonical(canonical, "tui.editor.undo")) { this.#applyUndo(); return; } @@ -1178,26 +1218,26 @@ export class Editor implements Component, Focusable { // Handle autocomplete special keys first (but don't block other input) if (this.#autocompleteState && this.#autocompleteList) { // Escape - cancel autocomplete - if (kb.matches(data, "tui.select.cancel")) { + if (kb.matchesCanonical(canonical, "tui.select.cancel")) { this.#cancelAutocomplete(true); return; } // Let the autocomplete list handle navigation and selection else if ( - kb.matches(data, "tui.select.up") || - kb.matches(data, "tui.select.down") || - kb.matches(data, "tui.select.pageUp") || - kb.matches(data, "tui.select.pageDown") || - kb.matches(data, "tui.input.submit") || + kb.matchesCanonical(canonical, "tui.select.up") || + kb.matchesCanonical(canonical, "tui.select.down") || + kb.matchesCanonical(canonical, "tui.select.pageUp") || + kb.matchesCanonical(canonical, "tui.select.pageDown") || + kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n" || - kb.matches(data, "tui.input.tab") + kb.matchesCanonical(canonical, "tui.input.tab") ) { // Only pass navigation keys to the list, not Enter/Tab (we handle those directly) if ( - kb.matches(data, "tui.select.up") || - kb.matches(data, "tui.select.down") || - kb.matches(data, "tui.select.pageUp") || - kb.matches(data, "tui.select.pageDown") + kb.matchesCanonical(canonical, "tui.select.up") || + kb.matchesCanonical(canonical, "tui.select.down") || + kb.matchesCanonical(canonical, "tui.select.pageUp") || + kb.matchesCanonical(canonical, "tui.select.pageDown") ) { this.#autocompleteList.handleInput(data); this.onAutocompleteUpdate?.(); @@ -1205,7 +1245,7 @@ export class Editor implements Component, Focusable { } // If Tab was pressed, always apply the selection - if (kb.matches(data, "tui.input.tab")) { + if (kb.matchesCanonical(canonical, "tui.input.tab")) { const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh // (destructive keys or paste can outrun the debounced update). @@ -1249,7 +1289,7 @@ export class Editor implements Component, Focusable { // If Enter was pressed on a submitted slash command (not an absolute-path // completion sharing the leading-slash prefix), apply and submit. if ( - (kb.matches(data, "tui.input.submit") || data === "\n") && + (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") && findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && this.#isInSubmittedSlashCommandContext() && !this.#selectedCompletionIsPath() @@ -1281,7 +1321,7 @@ export class Editor implements Component, Focusable { // Don't return - fall through to submission logic } // Otherwise, apply the completion without submitting the surrounding draft. - else if (kb.matches(data, "tui.input.submit") || data === "\n") { + else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") { const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh. const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; @@ -1321,37 +1361,37 @@ export class Editor implements Component, Focusable { } // Tab key - context-aware completion (but not when already autocompleting) - if (kb.matches(data, "tui.input.tab") && !this.#autocompleteState) { + if (kb.matchesCanonical(canonical, "tui.input.tab") && !this.#autocompleteState) { this.#handleTabCompletion(); return; } // Continue with rest of input handling // Delete to end of line - if (kb.matches(data, "tui.editor.deleteToLineEnd")) { + if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineEnd")) { this.#deleteToEndOfLine(); } // Delete to start of line - else if (kb.matches(data, "tui.editor.deleteToLineStart")) { + else if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineStart")) { this.#deleteToStartOfLine(); } // Delete word backward. Registry defaults cover ctrl+w, alt+backspace, // ctrl+backspace, and super+alt+backspace (Ghostty on macOS reports // Option+Backspace as super+alt — kitty mod 11, see #2064). - else if (kb.matches(data, "tui.editor.deleteWordBackward")) { + else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordBackward")) { this.#deleteWordBackwards(); } // Delete word forward. Registry defaults cover alt+d/alt+delete and their // super+alt variants for the same Ghostty quirk. - else if (kb.matches(data, "tui.editor.deleteWordForward")) { + else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordForward")) { this.#deleteWordForwards(); } // Yank from kill ring - else if (kb.matches(data, "tui.editor.yank")) { + else if (kb.matchesCanonical(canonical, "tui.editor.yank")) { this.#yankFromKillRing(); } // Yank-pop (cycle kill ring) - else if (kb.matches(data, "tui.editor.yankPop")) { + else if (kb.matchesCanonical(canonical, "tui.editor.yankPop")) { this.#yankPop(); } // Ctrl+A - Move to start of line @@ -1376,7 +1416,7 @@ export class Editor implements Component, Focusable { matchesKey(data, "ctrl+enter") || // Ctrl+Enter (Kitty/modifyOtherKeys, including lock bits/keypad Enter) data === "\x1b\r" || // Option+Enter in some terminals (legacy) data === "\x1b[13;2~" || // Shift+Enter in some terminals (legacy format) - kb.matches(data, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits) + kb.matchesCanonical(canonical, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits) (data.length > 1 && data.includes("\x1b") && data.includes("\r")) || (data === "\n" && data.length === 1) // Shift+Enter from iTerm2 mapping ) { @@ -1388,7 +1428,7 @@ export class Editor implements Component, Focusable { this.#addNewLine(); } // Plain Enter - submit (handles both legacy \r and Kitty protocol with lock bits) - else if (kb.matches(data, "tui.input.submit") || data === "\n") { + else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") { // If submit is disabled, do nothing if (this.disableSubmit) { return; @@ -1429,40 +1469,40 @@ export class Editor implements Component, Focusable { this.#submitValue(); } // Backspace (including Shift+Backspace) - else if (kb.matches(data, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) { + else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) { this.#handleBackspace(); } // Line navigation shortcuts (Home/End keys) - else if (kb.matches(data, "tui.editor.cursorLineStart")) { + else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineStart")) { this.#moveToLineStart(); - } else if (kb.matches(data, "tui.editor.cursorLineEnd")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineEnd")) { this.#moveToLineEnd(); } // Page navigation (PageUp/PageDown): page the editor viewport only. On a // short draft this is a no-op — it never steps prompt history (that stays // on Up/Down), so an idle empty editor swallows the keys instead of // surprising the user by loading the previous prompt (#4754). - else if (kb.matches(data, "tui.editor.pageUp")) { + else if (kb.matchesCanonical(canonical, "tui.editor.pageUp")) { this.#pageScroll(-1); - } else if (kb.matches(data, "tui.editor.pageDown")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.pageDown")) { this.#pageScroll(1); } // Forward delete (Fn+Backspace or Delete key, including Shift+Delete) - else if (kb.matches(data, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) { + else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) { this.#handleForwardDelete(); } // Word navigation (Option/Alt + Arrow or Ctrl + Arrow) - else if (kb.matches(data, "tui.editor.cursorWordLeft")) { + else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordLeft")) { // Word left this.#resetKillSequence(); this.#moveWordBackwards(); - } else if (kb.matches(data, "tui.editor.cursorWordRight")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordRight")) { // Word right this.#resetKillSequence(); this.#moveWordForwards(); } // Arrow keys - else if (kb.matches(data, "tui.editor.cursorUp")) { + else if (kb.matchesCanonical(canonical, "tui.editor.cursorUp")) { // Up - history navigation or cursor movement if (this.#isEditorEmpty()) { this.#navigateHistory(-1); // Start browsing history @@ -1474,7 +1514,7 @@ export class Editor implements Component, Focusable { } else { this.#moveCursor(-1, 0); // Cursor movement (within text or history entry) } - } else if (kb.matches(data, "tui.editor.cursorDown")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.cursorDown")) { // Down - history navigation or cursor movement if (this.#historyIndex > -1 && this.#isOnLastVisualLine()) { this.#navigateHistory(1); // Navigate to newer history entry or clear @@ -1484,10 +1524,10 @@ export class Editor implements Component, Focusable { } else { this.#moveCursor(1, 0); // Cursor movement (within text or history entry) } - } else if (kb.matches(data, "tui.editor.cursorRight")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.cursorRight")) { // Right this.#moveCursor(0, 1); - } else if (kb.matches(data, "tui.editor.cursorLeft")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.cursorLeft")) { // Left this.#moveCursor(0, -1); } @@ -1496,9 +1536,9 @@ export class Editor implements Component, Focusable { this.#insertCharacter(" "); } // Character jump mode triggers - else if (kb.matches(data, "tui.editor.jumpForward")) { + else if (kb.matchesCanonical(canonical, "tui.editor.jumpForward")) { this.#jumpMode = "forward"; - } else if (kb.matches(data, "tui.editor.jumpBackward")) { + } else if (kb.matchesCanonical(canonical, "tui.editor.jumpBackward")) { this.#jumpMode = "backward"; } // Printable keystrokes, including Kitty CSI-u text-producing sequences. @@ -1630,6 +1670,15 @@ export class Editor implements Component, Focusable { return this.#state.lines.join("\n"); } + /** Whether the buffer text equals `value`, without `getText()`'s full join — + * O(1) for the hot per-keystroke probes against short single-line values. */ + textEquals(value: string): boolean { + const lines = this.#state.lines; + if (lines.length === 1) return lines[0] === value; + if (value.indexOf("\n") === -1) return false; + return this.getText() === value; + } + #expandPasteMarkers(text: string): string { let result = text; for (const [pasteId, pasteContent] of this.#pastes) { diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index e9cec13a3..96d3d6dc7 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -1,5 +1,5 @@ import { LRUCache } from "lru-cache/raw"; -import { Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked"; +import { Lexer, Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked"; import { latexToBlock } from "../latex-block"; import { inlineMathSpanEnd, isBareMathEnvironment, latexToUnicode } from "../latex-to-unicode"; import type { SymbolTheme } from "../symbols"; @@ -492,12 +492,25 @@ const customHrExtension: TokenizerAndRendererExtension = { }, }; +// Leftmost-match scan replacing /\$|\\\(|\\\[/ in mathExtension.start — +// marked calls start() on the remaining source at every inline position, so +// the regex alternation showed up in CPU profiles (part of a ~4.3% start() +// tail). Three indexOf scans yield the identical leftmost index. +/** @internal exported for tests — must stay index-identical to the old regex scan. */ +export function mathStartIndex(src: string): number | undefined { + let best = src.indexOf("$"); + const paren = src.indexOf("\\("); + if (paren !== -1 && (best === -1 || paren < best)) best = paren; + const bracket = src.indexOf("\\["); + if (bracket !== -1 && (best === -1 || bracket < best)) best = bracket; + return best === -1 ? undefined : best; +} + const mathExtension: TokenizerAndRendererExtension = { name: "math", level: "inline", start(src) { - const m = /\$|\\\(|\\\[/.exec(src); - return m ? m.index : undefined; + return mathStartIndex(src); }, tokenizer(src) { if (src.startsWith("$$")) { @@ -614,14 +627,63 @@ const mathEnvBlockExtension: TokenizerAndRendererExtension = { // tokenizer at a valid start. Candidates at a legal boundary fall through // (return undefined) to marked's own autolink handling unchanged. const AUTOLINK_SCHEME_REGEX = /^(?:www\.|https?:\/\/|ftp:\/\/)/i; -const AUTOLINK_SCHEME_SCAN = /www\.|https?:\/\/|ftp:\/\//i; +// Case-insensitive scheme scan replacing /www\.|https?:\/\/|ftp:\/\//i in +// boundedAutolinkExtension.start — like mathStartIndex above, this runs on the +// remaining source at every inline position (part of a ~4.3% CPU start() scan +// tail in profiles). charCode-only: no allocation, no toLowerCase copies. +// `| 32` lower-cases ASCII letters; `.`/`:`/`/` are compared exactly, matching +// the regex's ASCII-only `i` semantics. charCodeAt past the end returns NaN, +// which fails every comparison, so no explicit bounds checks are needed. +function isAutolinkSchemeAt(src: string, i: number): boolean { + const c = src.charCodeAt(i) | 32; + if (c === 119 /* w */) { + // www. + return ( + (src.charCodeAt(i + 1) | 32) === 119 && + (src.charCodeAt(i + 2) | 32) === 119 && + src.charCodeAt(i + 3) === 46 /* . */ + ); + } + if (c === 104 /* h */) { + // http:// | https:// + if ( + (src.charCodeAt(i + 1) | 32) !== 116 /* t */ || + (src.charCodeAt(i + 2) | 32) !== 116 /* t */ || + (src.charCodeAt(i + 3) | 32) !== 112 /* p */ + ) { + return false; + } + let j = i + 4; + if ((src.charCodeAt(j) | 32) === 115 /* s */) j++; + return src.charCodeAt(j) === 58 /* : */ && src.charCodeAt(j + 1) === 47 /* / */ && src.charCodeAt(j + 2) === 47; + } + if (c === 102 /* f */) { + // ftp:// + return ( + (src.charCodeAt(i + 1) | 32) === 116 /* t */ && + (src.charCodeAt(i + 2) | 32) === 112 /* p */ && + src.charCodeAt(i + 3) === 58 /* : */ && + src.charCodeAt(i + 4) === 47 /* / */ && + src.charCodeAt(i + 5) === 47 /* / */ + ); + } + return false; +} + +/** @internal exported for tests — must stay index-identical to the old regex scan. */ +export function autolinkSchemeScanIndex(src: string): number | undefined { + for (let i = 0; i < src.length; i++) { + const c = src.charCodeAt(i) | 32; + if ((c === 119 || c === 104 || c === 102) && isAutolinkSchemeAt(src, i)) return i; + } + return undefined; +} const VALID_AUTOLINK_LEFT_BOUNDARY = /[\s*_~(]/; const boundedAutolinkExtension: TokenizerAndRendererExtension = { name: "boundedAutolink", level: "inline", start(src) { - const m = AUTOLINK_SCHEME_SCAN.exec(src); - return m ? m.index : undefined; + return autolinkSchemeScanIndex(src); }, tokenizer(src, tokens) { const match = AUTOLINK_SCHEME_REGEX.exec(src); @@ -639,6 +701,66 @@ markdownParser.use({ extensions: [customHrExtension, mathBlockExtension, mathEnvBlockExtension, mathExtension, boundedAutolinkExtension], }); +// --------------------------------------------------------------------------- +// GFM `url` tokenizer gate +// --------------------------------------------------------------------------- +// marked tries the bundled GFM `url` tokenizer at every inline tokenization +// step, and its regex is expensive to FAIL: the email alternative +// `^[A-Za-z0-9._+-]+(@)…` linearly consumes an identifier run, then backtracks +// it one character at a time when no `@` follows. A 71414-sample / 1ms CPU +// profile of the TUI put 73.3% of total CPU (74.9s of a 102s capture) inside +// this single regex. The override below runs an O(bounded) charCode gate first +// and only falls through to the built-in tokenizer — by returning `false`, +// marked's tokenizer-override fallback contract — when a match is possible. +// +// Conservativeness argument. The built-in rule (no flags) is +// /^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.?)+[^\s<]* +// |^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/ +// Both alternatives are anchored, so any match constrains the head of src: +// • Branch 1 requires src to start with `http://`, `https://`, `ftp://` +// (scheme letters in any case) or lowercase `www.`. The gate accepts all of +// these via isAutolinkSchemeAt(src, 0); it also over-accepts `WWW.`, a +// harmless false positive (the built-in regex simply fails to match). +// • Branch 2 requires src to start with one-or-more chars from +// `[A-Za-z0-9._+-]` immediately followed by `@`. The gate scans that exact +// class: if the run ends within URL_GATE_EMAIL_SCAN_LIMIT chars it accepts +// iff the terminator is `@`; a run reaching the limit is accepted +// unconditionally. Every src branch 2 can match is therefore accepted — +// the gate never rejects a src the built-in regex would match. +const URL_GATE_EMAIL_SCAN_LIMIT = 320; + +/** @internal exported for tests — must never return false for a src the built-in url regex matches. */ +export function urlTokenPossible(src: string): boolean { + if (isAutolinkSchemeAt(src, 0)) return true; + let i = 0; + while (i < URL_GATE_EMAIL_SCAN_LIMIT) { + const c = src.charCodeAt(i); + const isLocalChar = + (c >= 97 && c <= 122) /* a-z */ || + (c >= 65 && c <= 90) /* A-Z */ || + (c >= 48 && c <= 57) /* 0-9 */ || + c === 46 /* . */ || + c === 95 /* _ */ || + c === 43 /* + */ || + c === 45; /* - */ + if (!isLocalChar) break; + i++; + } + if (i === 0) return false; + if (i >= URL_GATE_EMAIL_SCAN_LIMIT) return true; // over-long run: give up conservatively + return src.charCodeAt(i) === 64 /* @ */; +} + +markdownParser.use({ + tokenizer: { + url(src: string): Tokens.Link | undefined | false { + // `false` → marked falls back to the built-in `url` tokenizer; + // `undefined` → no url token here, built-in never runs. + return urlTokenPossible(src) ? false : undefined; + }, + }, +}); + // --------------------------------------------------------------------------- // Module-level LRU render cache // --------------------------------------------------------------------------- @@ -649,8 +771,8 @@ markdownParser.use({ // (Rust FFI) work for content/layout combinations already seen this session. const RENDER_CACHE_MAX = 256; // sane cap: ~256 distinct message × width combos -const RENDER_CACHE_MAX_SIZE = 512 * 1024; -const RENDER_CACHE_MAX_ENTRY_SIZE = 32 * 1024; +const RENDER_CACHE_MAX_SIZE = 4 * 1024 * 1024; +const RENDER_CACHE_MAX_ENTRY_SIZE = 256 * 1024; const EMPTY_RENDER_LINES: readonly string[] = []; interface RenderCacheEntry { @@ -684,6 +806,179 @@ function renderCacheEntrySize(entry: RenderCacheEntry): number { // over-matching is safe (it only costs the fast path), under-matching is not. const HAS_REF_DEF = /^ {0,3}\[(?:\\.|[^\]\\])+\]:/m; +// marked's list tokenizer (Tokenizer.list, marked v18) continues a list across +// blank lines only when the remaining source matches +// `listItemRegex(marker)` = `^( {0,3}${marker})((?:[\t ][^\n]*)?(?:\n|$))`, +// where `marker` is the exact bullet char for unordered lists (`\${char}`) or +// 1-9 digits plus the exact delimiter for ordered lists (`\d{1,9}\${delim}`). +// The marker is derived from the list's FIRST item (`n = t[1].trim()`), which +// sits at the start of a top-level list token's raw: +const LIST_MARKER_RE = /^ {0,3}(?:([*+-])|\d{1,9}([.)]))/; + +// Streaming-freeze equivalence invariant: lex(prefix) ++ lex(tail) must equal +// lex(full text) — for the CURRENT text and for every append-only extension of +// it, because a frozen prefix is sticky (it keeps being reused while the text +// grows). At a blank-line (`\n\n`) cut directly after a top-level `list` +// token, the only construct that can straddle the cut is a continuation item +// of that list: marked consumed the blank line into the last item's raw and +// re-ran `listItemRegex` at exactly `tailStart`, merging a same-marker item +// into one renumbered loose list. The cut is safe only when that regex can +// NEVER match at `tailStart`, no matter what is appended later. +// +// Append-only growth means existing characters are immutable while new ones +// may appear after them, so "closed" may only be concluded from a present +// character that contradicts every possible continuation (e.g. tail "1x" can +// never grow into an ordered item, but tail "1" can become "1. c"). Running +// out of text mid-marker therefore answers "may continue". +// +// Returns true when the tail could still continue the list (or the list's +// marker is unrecognizable) — the conservative "don't freeze" answer. marked +// may break the list anyway when the matching line is also an hr (`- - -`); +// treating that as "may continue" merely skips a freeze, never corrupts one. +function listMayContinueAt(text: string, tailStart: number, listRaw: string): boolean { + const marker = LIST_MARKER_RE.exec(listRaw); + if (marker === null) return true; // unrecognized list shape — stay conservative + const n = text.length; + let i = tailStart; + // `listItemRegex` allows up to 3 leading spaces (the caller's next-char + // guard rejects whitespace at the final cut, but mirror the rule exactly). + while (i < n && i - tailStart < 3 && text.charCodeAt(i) === 0x20 /* space */) i++; + if (i >= n) return true; + const bullet = marker[1]; + if (bullet !== undefined) { + if (text[i] !== bullet) return false; // wrong marker char — closed forever + i++; + } else { + // Ordered: 1-9 digits, then the same `.`/`)` delimiter. + let digits = 0; + while (i < n && digits < 10) { + const c = text.charCodeAt(i); + if (c < 0x30 /* 0 */ || c > 0x39 /* 9 */) break; + digits++; + i++; + } + if (digits === 0 || digits > 9) return false; // no digit run / too long — closed forever + if (i >= n) return true; // delimiter (or more digits) may still arrive + if (text[i] !== marker[2]) return false; // wrong delimiter — closed forever + i++; + } + // After the marker: `(?:[\t ][^\n]*)?(?:\n|$)` — tab/space + anything, a + // bare newline, or end-of-input (which appends can still extend). + if (i >= n) return true; + const after = text.charCodeAt(i); + return after === 0x20 /* space */ || after === 0x09 /* tab */ || after === 0x0a /* \n */; +} + +const NO_BLOCK_BOUNDARY = { end: 0, count: 0 } as const; + +/** + * Offset just past the last token in `tokens` that closes a block on a hard + * `"\n\n"` break, together with the number of tokens up to and including it. + * `count === 0` means the run holds no usable boundary. + * + * `base` is where `tokens[0]` starts inside `text`. A boundary qualifies only + * when splitting there is invisible to the lexer, i.e. `lex(head) ++ lex(tail) + * === lex(text)`: + * - The break must sit inside `text`. At end-of-text the next character is + * unknown (and, while streaming, may still arrive), so the cut is deferred. + * - The next character must start real block content. Whitespace means the + * block separator straddles the cut — e.g. a fence followed by + * `"\n\n\n- list"` — and the two lexes desync. + * - A preceding `list` must be provably closed: CommonMark lets a same-marker + * item continue the list across the blank line, and marked merges both into + * one renumbered loose list (`listMayContinueAt`). + */ +function stableBlockBoundary(text: string, base: number, tokens: Token[]): { end: number; count: number } { + let pos = base; + let end = 0; + let count = 0; + for (let i = 0; i < tokens.length; i++) { + const raw = tokens[i].raw; + const tokenEnd = pos + raw.length; + if (raw.endsWith("\n\n")) { + const prev = i > 0 ? tokens[i - 1] : undefined; + if (prev === undefined || prev.type !== "list" || !listMayContinueAt(text, tokenEnd, prev.raw)) { + end = tokenEnd; + count = i + 1; + } + } + pos = tokenEnd; + } + if (count === 0 || end >= text.length) return NO_BLOCK_BOUNDARY; + const next = text.charCodeAt(end); + if (next === 0x20 /* space */ || next === 0x0a /* \n */) return NO_BLOCK_BOUNDARY; + return { end, count }; +} + +// Bun's regex engine skips the start-anchor optimization for several of marked's +// block rules — `hr`, `lheading`, `table` and `html` are `^`-anchored +// alternations of quantified branches — so each failing `exec` rescans the whole +// remaining source instead of stopping at offset 0. Lexing is then quadratic in +// document length: an 800 KB message costs ~41 s under Bun where Node/V8 needs +// ~60 ms, and it runs on the render path, freezing the UI. Bounded windows keep +// every scan short and restore linear behavior (~0.7 s for that same message). +const LEX_WINDOW_BYTES = 2 * 1024; +// Under this size a single pass beats probing for window boundaries; the +// crossover measured on pathological Markdown sits around 16 KB. +const WINDOWED_LEX_MIN_BYTES = 16 * 1024; + +/** + * Lex `text` in bounded windows, producing the exact token stream + * `markdownParser.lexer(text)` would. + * + * Window cuts come from marked itself: a throwaway BLOCK-ONLY probe lex of the + * window reports its last stable block boundary ({@link stableBlockBoundary}) + * and only that confirmed segment is handed to the real lexer; a window + * holding no boundary doubles until it finds one or reaches the end. Probes + * never run inline tokenization (their inlineQueue is discarded) — a boundary + * is a property of block structure alone, and probe inline passes were the + * dominant cost of an earlier revision. Block tokenization runs per window + * while inline tokenization is deferred to the end — mirroring `Lexer.lex` — + * so a `[label]: dest` definition anywhere in the document still resolves for + * every inline span. + * + * A boundary requires some top-level token whose raw ends in `"\n\n"`, so a + * window that contains no blank line cannot cut: each round starts at the next + * `"\n\n"` (skipping straight to the end when there is none — e.g. a tail + * that is one long tight list) instead of probing sizes that cannot succeed. + */ +function lexWindowed(text: string): Token[] { + const lexer = new Lexer(markdownParser.defaults); + let offset = 0; + while (offset < text.length) { + let segment = ""; + const nextBlank = text.indexOf("\n\n", offset); + if (nextBlank === -1) { + segment = text.slice(offset); + } else { + const minSize = Math.max(LEX_WINDOW_BYTES, nextBlank + 2 - offset); + for (let size = minSize; segment.length === 0; size *= 2) { + if (offset + size >= text.length) { + segment = text.slice(offset); + break; + } + const probe = new Lexer(markdownParser.defaults); + probe.blockTokens(text.slice(offset, offset + size), probe.tokens); + const boundary = stableBlockBoundary(text, offset, probe.tokens); + if (boundary.count > 0) segment = text.slice(offset, boundary.end); + } + } + lexer.blockTokens(segment, lexer.tokens); + offset += segment.length; + } + for (const queued of lexer.inlineQueue) lexer.inlineTokens(queued.src, queued.tokens); + lexer.inlineQueue = []; + return lexer.tokens; +} + +/** Lex a whole document, windowing anything large enough for the quadratic scan to bite. */ +function lexDocument(text: string): Token[] { + // A CR shifts every `raw` span (marked normalizes CRLF before tokenizing), so + // window offsets would address the wrong characters — lex those in one pass. + if (text.length < WINDOWED_LEX_MIN_BYTES || text.includes("\r")) return markdownParser.lexer(text); + return lexWindowed(text); +} + /** Drop all L2 cache entries. Call on theme change to prevent stale styled output. */ export function clearRenderCache(): void { renderCache.clear(); @@ -1219,12 +1514,12 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ text.length > prefix.length && text.startsWith(prefix) ) { - const tailTokens = markdownParser.lexer(text.slice(prefix.length)); + const tailTokens = lexDocument(text.slice(prefix.length)); const tokens = [...prefixTokens, ...tailTokens]; this.#freezeStablePrefix(text, tokens, { preserveExisting: true }); return tokens; } - const tokens = markdownParser.lexer(text); + const tokens = lexDocument(text); if (canStream) { this.#freezeStablePrefix(text, tokens, { preserveExisting: false }); } else { @@ -1241,36 +1536,11 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ // reference definitions, so each token's `raw` is a verbatim slice of `text` // and the summed offsets address `text` exactly. #freezeStablePrefix(text: string, tokens: Token[], opts: { preserveExisting: boolean }): void { - let pos = 0; - let frozenEnd = 0; - let frozenCount = 0; - for (let i = 0; i < tokens.length; i++) { - const raw = tokens[i].raw; - const end = pos + raw.length; - // A `space` token ending in "\n\n" closes the preceding block, but a - // `list` before it can still be extended by a following same-marker - // item across the blank line (CommonMark loose-list continuation), - // which marked merges into one renumbered loose list. Freezing across - // such a cut would keep the lists separate. Never freeze right after a - // list — it stays in the re-lexed tail. - if (raw.endsWith("\n\n") && tokens[i - 1]?.type !== "list") { - frozenEnd = end; - frozenCount = i + 1; - } - pos = end; - } - // Freeze only when the tail begins with real block content. If the next - // char is whitespace (an extra blank line, or an indented continuation), - // the block separator straddles the cut and lex(prefix)++lex(tail) would - // desync from a full lex — e.g. a fence followed by "\n\n\n- list". When - // frozenEnd is at end-of-text the next char is unknown, so defer. - if (frozenCount > 0 && frozenEnd < text.length) { - const next = text.charCodeAt(frozenEnd); - if (next !== 0x20 /* space */ && next !== 0x0a /* \n */) { - this.#streamPrefixText = text.slice(0, frozenEnd); - this.#streamPrefixTokens = tokens.slice(0, frozenCount); - return; - } + const frozen = stableBlockBoundary(text, 0, tokens); + if (frozen.count > 0) { + this.#streamPrefixText = text.slice(0, frozen.end); + this.#streamPrefixTokens = tokens.slice(0, frozen.count); + return; } if (!opts.preserveExisting) { diff --git a/packages/tui/src/keybindings.ts b/packages/tui/src/keybindings.ts index 86bb32617..9d9cfcfaf 100644 --- a/packages/tui/src/keybindings.ts +++ b/packages/tui/src/keybindings.ts @@ -288,8 +288,17 @@ export class KeybindingsManager { matches(data: string, keybinding: Keybinding): boolean { const parsed = parseKey(data); if (parsed === undefined) return false; - const matchKeys = this.#matchKeysById.get(keybinding); - return matchKeys?.has(canonicalKeyId(parsed)) ?? false; + return this.matchesCanonical(canonicalKeyId(parsed), keybinding); + } + + /** + * Set-lookup variant of {@link matches} for hot input paths: the caller + * parses `data` once (`parseKey` + `canonicalKeyId`) and probes many + * bindings without re-parsing the raw sequence per probe. + */ + matchesCanonical(canonical: string | undefined, keybinding: Keybinding): boolean { + if (canonical === undefined) return false; + return this.#matchKeysById.get(keybinding)?.has(canonical) ?? false; } getKeys(keybinding: Keybinding): KeyId[] { diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index 855e2dfee..093fe7d1d 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -286,7 +286,7 @@ export function emergencyTerminalRestore(): void { terminal.write("\x1b[?1049l"); altScreenActive = false; } - terminal.showCursor(); + terminal.showCursor(true); } else if (terminalEverStarted && !isTerminalHeadless()) { // Blind restore only if we know a terminal was started but lost track of it // This avoids writing escape sequences for non-TUI commands (grep, commit, etc.) @@ -362,9 +362,11 @@ export interface Terminal { // Cursor positioning (relative to current position) moveBy(lines: number): void; // Move cursor up (negative) or down (positive) by N lines - // Cursor visibility - hideCursor(): void; // Hide the cursor - showCursor(): void; // Show the cursor + // Cursor visibility. Same-state calls are deduped against the visibility + // last written to the terminal; pass force=true to write unconditionally + // (crash/exit restore paths). + hideCursor(force?: boolean): void; // Hide the cursor + showCursor(force?: boolean): void; // Show the cursor // Clear operations clearLine(): void; // Clear current line @@ -478,6 +480,12 @@ export class ProcessTerminal implements Terminal { this.#markTerminalDisconnected("stdin failed", err); }; #dead = false; + // Last cursor visibility written to the terminal, sniffed from every + // outgoing sequence (frame buffers embed their own ?25h/?25l), so + // hideCursor()/showCursor() can skip same-state writes. `undefined` = + // unknown (fresh start, resize, or an alt-screen switch newer than the + // last cursor sequence — some hosts keep DECTCEM per buffer). + #cursorVisible: boolean | undefined; // Captured at construction and re-read at start(): when true, every real // terminal side effect (writes, probes, raw mode, SIGWINCH, timers) is // suppressed. Defaults on under `bun test` — see isTerminalHeadless(). @@ -575,6 +583,8 @@ export class ProcessTerminal implements Terminal { this.#inputHandler = onInput; this.#resizeHandler = onResize; this.#disconnectHandler = onDisconnect; + // The host terminal's cursor visibility is unknown until we write it. + this.#cursorVisible = undefined; // Headless (tests): suppress every real-terminal side effect. Skip raw // mode, stdin listeners, capability probes, SIGWINCH, and emergency-restore @@ -620,6 +630,9 @@ export class ProcessTerminal implements Terminal { // dimensions before firing `resize`, so it is authoritative for geometry: // reconcile any stale cached DEC 2048 report before notifying the renderer. this.#stdoutResizeListener = () => { + // Conservative: some hosts reset modes across a resize/reattach, so + // re-establish cursor visibility on the next explicit call. + this.#cursorVisible = undefined; this.#reconcileInBandGeometryOnResize(); this.#resizeHandler?.(); }; @@ -1474,6 +1487,9 @@ export class ProcessTerminal implements Terminal { } this.#stdoutErrorCleanup?.(); this.#stdoutErrorCleanup = undefined; + // After stop() the terminal is shared with other writers; visibility + // tracking is only meaningful while this instance owns the TTY. + this.#cursorVisible = undefined; } #ensureStdoutErrorHandler(): void { @@ -1527,6 +1543,7 @@ export class ProcessTerminal implements Terminal { // files). They serve no purpose there and would surface as visible noise. if (!process.stdout.isTTY) return; this.#ensureStdoutErrorHandler(); + this.#trackCursorVisibility(data); // A console-sharing child process may have flipped the console codepage // away from UTF-8; repair it before any bytes hit WriteFile so no frame // is ever translated through an OEM codepage. See ensureWindowsConsoleUtf8. @@ -1578,14 +1595,37 @@ export class ProcessTerminal implements Terminal { // lines === 0: no movement } - hideCursor(): void { + hideCursor(force = false): void { + if (!force && this.#cursorVisible === false) return; this.#safeWrite("\x1b[?25l"); } - showCursor(): void { + showCursor(force = false): void { + if (!force && this.#cursorVisible === true) return; this.#safeWrite("\x1b[?25h"); } + /** + * Sniff outgoing data for the last cursor-visibility change so the tracked + * state stays correct for sequences embedded in frame buffers + * (TUI#cursorControlSequence appends ?25h/?25l inside the paint write). An + * alt-screen switch (DECSET/DECRST 1049) newer than the last cursor + * sequence resets tracking to unknown: some hosts keep DECTCEM per buffer. + */ + #trackCursorVisibility(data: string): void { + let idx = data.lastIndexOf("\x1b[?25"); + while (idx !== -1) { + const final = data.charCodeAt(idx + 5); + if (final === 0x68 /* h */ || final === 0x6c /* l */) break; + idx = idx === 0 ? -1 : data.lastIndexOf("\x1b[?25", idx - 1); + } + if (data.lastIndexOf("\x1b[?1049") > idx) { + this.#cursorVisible = undefined; + return; + } + if (idx !== -1) this.#cursorVisible = data.charCodeAt(idx + 5) === 0x68; + } + clearLine(): void { this.#safeWrite("\x1b[K"); } diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index f9e4fb744..7b309b8af 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1875,7 +1875,9 @@ export class TUI extends Container { this.terminal.write(targetRow <= viewportBottom ? "\r" : "\r\n"); } - this.terminal.showCursor(); + // Force: the parent shell needs the cursor back regardless of what the + // terminal-level dedupe believes was last written. + this.terminal.showCursor(true); this.#forgetHardwareCursorState(); this.terminal.stop(); } diff --git a/packages/tui/test/cursor-visibility-dedupe.test.ts b/packages/tui/test/cursor-visibility-dedupe.test.ts new file mode 100644 index 000000000..624db5e85 --- /dev/null +++ b/packages/tui/test/cursor-visibility-dedupe.test.ts @@ -0,0 +1,148 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { ProcessTerminal } from "@oh-my-pi/pi-tui/terminal"; +import { setTerminalHeadless } from "@oh-my-pi/pi-utils"; + +// ProcessTerminal dedupes cursor-visibility writes: hideCursor()/showCursor() +// skip the ?25l/?25h escape when the terminal already holds that state. The +// tracked state is sniffed from every outgoing write, so cursor sequences +// embedded in frame buffers (TUI appends ?25h/?25l inside the paint write) +// keep it in sync, and an alt-screen switch resets it to unknown. Crash/exit +// restore paths pass force=true and must always write. + +const HIDE = "\x1b[?25l"; +const SHOW = "\x1b[?25h"; + +const stdinIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "isTTY"); +const stdoutIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "isTTY"); +const stdinSetRawModeDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "setRawMode"); +let previousHeadless = false; + +function restoreProperty(target: object, key: string, descriptor: PropertyDescriptor | undefined): void { + if (descriptor) { + Object.defineProperty(target, key, descriptor); + return; + } + delete (target as Record)[key]; +} + +function startCapturedTerminal() { + const writes: string[] = []; + Object.defineProperty(process.stdin, "isTTY", { value: true, configurable: true }); + Object.defineProperty(process.stdout, "isTTY", { value: true, configurable: true }); + Object.defineProperty(process.stdin, "setRawMode", { value: vi.fn(), configurable: true }); + vi.spyOn(process, "kill").mockReturnValue(true); + vi.spyOn(process.stdin, "resume").mockImplementation(() => process.stdin); + vi.spyOn(process.stdin, "pause").mockImplementation(() => process.stdin); + vi.spyOn(process.stdin, "setEncoding").mockImplementation(() => process.stdin); + vi.spyOn(process.stdout, "write").mockImplementation(chunk => { + writes.push(String(chunk)); + return true; + }); + + const terminal = new ProcessTerminal(); + terminal.start( + () => {}, + () => {}, + ); + writes.length = 0; + return { terminal, writes }; +} + +describe("ProcessTerminal cursor-visibility dedupe", () => { + beforeEach(() => { + previousHeadless = setTerminalHeadless(false); + }); + + afterEach(() => { + setTerminalHeadless(previousHeadless); + vi.restoreAllMocks(); + restoreProperty(process.stdin, "isTTY", stdinIsTtyDescriptor); + restoreProperty(process.stdout, "isTTY", stdoutIsTtyDescriptor); + restoreProperty(process.stdin, "setRawMode", stdinSetRawModeDescriptor); + }); + + it("writes each visibility change once and skips same-state repeats", () => { + const { terminal, writes } = startCapturedTerminal(); + + terminal.hideCursor(); + terminal.hideCursor(); + terminal.hideCursor(); + expect(writes).toEqual([HIDE]); + + terminal.showCursor(); + terminal.showCursor(); + expect(writes).toEqual([HIDE, SHOW]); + + terminal.hideCursor(); + expect(writes).toEqual([HIDE, SHOW, HIDE]); + terminal.stop(); + }); + + it("tracks cursor sequences embedded in frame writes", () => { + const { terminal, writes } = startCapturedTerminal(); + + // A paint that repositions the hardware cursor ends by showing it. + terminal.write(`\x1b[2Bframe content\x1b[5G${SHOW}\x1b[?2026l`); + writes.length = 0; + + terminal.showCursor(); // already visible per the frame write + expect(writes).toEqual([]); + + terminal.hideCursor(); // state change: must write + expect(writes).toEqual([HIDE]); + terminal.stop(); + }); + + it("honors the last of multiple cursor sequences in one write", () => { + const { terminal, writes } = startCapturedTerminal(); + + terminal.write(`${SHOW}overlay paint${HIDE}`); + writes.length = 0; + + terminal.hideCursor(); + expect(writes).toEqual([]); + terminal.showCursor(); + expect(writes).toEqual([SHOW]); + terminal.stop(); + }); + + it("force-writes regardless of tracked state (crash/exit restore contract)", () => { + const { terminal, writes } = startCapturedTerminal(); + + terminal.showCursor(); + terminal.showCursor(true); + terminal.showCursor(true); + expect(writes).toEqual([SHOW, SHOW, SHOW]); + + terminal.hideCursor(); + terminal.hideCursor(true); + expect(writes).toEqual([SHOW, SHOW, SHOW, HIDE, HIDE]); + terminal.stop(); + }); + + it("resets tracking to unknown when an alt-screen switch follows the last cursor sequence", () => { + const { terminal, writes } = startCapturedTerminal(); + + terminal.hideCursor(); + // Alt-screen enter after the hide: some hosts keep DECTCEM per buffer, + // so the tracked state is no longer trustworthy. + terminal.write("\x1b[?1049h\x1b[2J"); + writes.length = 0; + + terminal.hideCursor(); + expect(writes).toEqual([HIDE]); + terminal.stop(); + }); + + it("does not confuse other private modes with cursor visibility", () => { + const { terminal, writes } = startCapturedTerminal(); + + terminal.hideCursor(); + writes.length = 0; + // Neither DECRQM on mode 25 nor unrelated ?25xx modes change visibility. + terminal.write("\x1b[?25$p\x1b[?2026h"); + terminal.hideCursor(); + expect(writes).toEqual(["\x1b[?25$p\x1b[?2026h"]); + terminal.stop(); + }); +}); diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index 94c61869c..3bedd83b9 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -2396,6 +2396,110 @@ describe("Editor component", () => { }); }); + describe("Bulk input fast path and paste iteration", () => { + it("produces identical state for a chunked paste and a single-sequence paste", () => { + const content = "alpha beta\ngamma delta\nepsilon"; + const single = new Editor(defaultEditorTheme); + single.handleInput(`\x1b[200~${content}\x1b[201~`); + + const chunked = new Editor(defaultEditorTheme); + chunked.handleInput("\x1b[200~"); + for (const ch of content) chunked.handleInput(ch); + chunked.handleInput("\x1b[201~"); + + expect(chunked.getText()).toBe(single.getText()); + expect(chunked.getCursor()).toEqual(single.getCursor()); + }); + + it("normalizes CRLF identically for single and chunked paste delivery", () => { + const single = new Editor(defaultEditorTheme); + single.handleInput("\x1b[200~one\r\ntwo\rthree\x1b[201~"); + + const chunked = new Editor(defaultEditorTheme); + chunked.handleInput("\x1b[200~one\r"); + chunked.handleInput("\ntwo"); + chunked.handleInput("\rthree\x1b[201~"); + + expect(single.getText()).toBe("one\ntwo\nthree"); + expect(chunked.getText()).toBe(single.getText()); + expect(chunked.getCursor()).toEqual(single.getCursor()); + }); + + it("processes paste remainders iteratively, applying every trailing paste and keystroke", () => { + const editor = new Editor(defaultEditorTheme); + // One read carrying two complete pastes plus trailing typed text: the + // remainder after each paste loops back through input handling. + editor.handleInput("\x1b[200~ab\x1b[201~\x1b[200~cd\x1b[201~ef"); + expect(editor.getText()).toBe("abcdef"); + expect(editor.getCursor()).toEqual({ line: 0, col: 6 }); + }); + + it("handles a long train of pastes in one read without recursing per remainder", () => { + const editor = new Editor(defaultEditorTheme); + editor.handleInput("\x1b[200~x\x1b[201~".repeat(2000)); + expect(editor.getText()).toBe("x".repeat(2000)); + }); + + it("inserts a plain printable run identically to per-scalar delivery", () => { + const run = "The quick brown fox 123 -_. naïve 😀 path"; + const bulk = new Editor(defaultEditorTheme); + bulk.handleInput(run); + + const perChar = new Editor(defaultEditorTheme); + for (const ch of run) perChar.handleInput(ch); + + expect(bulk.getText()).toBe(perChar.getText()); + expect(bulk.getCursor()).toEqual(perChar.getCursor()); + }); + + it("keeps escape sequences interleaved with printable runs on the dispatch path", () => { + const editor = new Editor(defaultEditorTheme); + editor.handleInput("abc"); + editor.handleInput("\x1b[D"); // Left + editor.handleInput("XY"); // bulk run lands before "c" + expect(editor.getText()).toBe("abXYc"); + expect(editor.getCursor()).toEqual({ line: 0, col: 4 }); + }); + + it("opens @ autocomplete when the trigger arrives inside a bulk printable run", async () => { + const editor = new Editor(defaultEditorTheme); + const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers(); + editor.setAutocompleteProvider({ + async getSuggestions() { + return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" }; + }, + applyCompletion(lines, cursorLine, cursorCol) { + return { lines, cursorLine, cursorCol }; + }, + }); + editor.onAutocompleteUpdate = resolveAutocompleteUpdated; + + editor.handleInput("see @sr"); // one bulk run ending in an @-token + + await autocompleteUpdated; + expect(editor.isShowingAutocomplete()).toBe(true); + }); + + it("opens @ autocomplete after a bracketed paste ending in a trigger token", async () => { + const editor = new Editor(defaultEditorTheme); + const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers(); + editor.setAutocompleteProvider({ + async getSuggestions() { + return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" }; + }, + applyCompletion(lines, cursorLine, cursorCol) { + return { lines, cursorLine, cursorCol }; + }, + }); + editor.onAutocompleteUpdate = resolveAutocompleteUpdated; + + editor.handleInput("\x1b[200~see @sr\x1b[201~"); + + await autocompleteUpdated; + expect(editor.isShowingAutocomplete()).toBe(true); + }); + }); + describe("Korean NFC paste normalization", () => { // macOS Finder drag-drops/Copy-As-Pathname emit Korean filenames as // NFD (decomposed) — e.g. `화` becomes `ᄒ`(U+1112) + `ᅪ`(U+116A). diff --git a/packages/tui/test/markdown-incremental-lex.test.ts b/packages/tui/test/markdown-incremental-lex.test.ts index fb46e9301..dc4d54f3c 100644 --- a/packages/tui/test/markdown-incremental-lex.test.ts +++ b/packages/tui/test/markdown-incremental-lex.test.ts @@ -243,4 +243,87 @@ describe("Markdown incremental streaming lex (E2)", () => { expect(streamLines).toEqual(renderCold(crlf.slice(0, len), 60)); } }); + + // Closed-list lookahead: a "\n\n" boundary directly after a list token is + // freezable iff the tail cannot start a continuation item of that list + // (same bullet char, or 1-9 digits + same delimiter — marked's + // listItemRegex). These corpora cross list/non-list and + // list/incompatible-list boundaries; the divergence (and the freeze + // opportunity) is phase-sensitive, so each runs at step=1 and the + // production reveal granularity (step=3). + it("bullet list followed by a paragraph grows byte-identically", () => { + const doc = + "- alpha item with words\n- beta item with words\n- gamma item\n\n" + + "Closing paragraph that keeps streaming additional words to the end."; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + assertIdenticalGrowthTransient(doc, 60, 3); + }); + + it("bullet list followed by a different-marker list stays two lists", () => { + const doc = "- alpha\n- beta\n\n* starred one\n* starred two\n\n+ plus one\n+ plus two"; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + }); + + it("ordered list followed by a paren-delimited list stays two lists", () => { + const doc = "1. dot one\n2. dot two\n\n1) paren one\n2) paren two"; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + assertIdenticalGrowthTransient(doc, 60, 3); + }); + + it("list followed by blockquote grows byte-identically", () => { + const doc = "- alpha\n- beta\n\n> quoted line one with words\n> quoted line two here"; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + }); + + it("list followed by heading grows byte-identically", () => { + const doc = "1. one\n2. two\n\n# Heading after the list\n\nTail prose keeps going on."; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + }); + + it("list followed by fenced code grows byte-identically", () => { + const doc = "- alpha\n- beta\n\n```ts\nconst x = compute(a, b);\nreturn x;\n```\n\ntail text"; + assertIdenticalGrowth(doc, 60, 1); + assertIdenticalGrowth(doc, 60, 3); + }); + + it("a same-marker list across a blank line still merges while growing", () => { + const bullets = "- a\n- b\n\n- c\n- d"; + assertIdenticalGrowth(bullets, 60, 1); + assertIdenticalGrowth(bullets, 60, 3); + }); + + it("a list closed by a paragraph freezes at the boundary (streaming perf gate)", () => { + // The lookahead must actually fire here: the tail after the blank line + // is a paragraph, which cannot continue a `-` list, so the rendered + // list rows become settled (frozen prefix) on the transient path. + const doc = "- alpha\n- beta\n- gamma\n\nClosing paragraph after the list keeps going."; + const streaming = new Markdown("", 0, 0, THEME); + streaming.transientRenderCache = true; + clearRenderCache(); + streaming.setText(doc); + const streamLines = streaming.render(60); + expect(streamLines).toEqual(renderCold(doc, 60)); + expect(streaming.getLastRenderSettledRows()).toBeGreaterThan(0); + }); + + it("a document that is one still-growing list never freezes mid-list", () => { + // No (b)-style intra-list freezing shipped: loose/tight and ordered + // renumbering are whole-list properties, so no prefix of an open list + // is byte-stable. Settled rows must stay 0 for a pure-list document. + const doc = "- one two three\n- four five six\n\n- seven eight nine"; + const streaming = new Markdown("", 0, 0, THEME); + streaming.transientRenderCache = true; + for (let len = 1; len <= doc.length; len += 1) { + clearRenderCache(); + streaming.setText(doc.slice(0, len)); + const streamLines = streaming.render(60); + expect(streamLines).toEqual(renderCold(doc.slice(0, len), 60)); + expect(streaming.getLastRenderSettledRows()).toBe(0); + } + }); }); diff --git a/packages/tui/test/markdown.test.ts b/packages/tui/test/markdown.test.ts index 23e90c998..ce5e2d514 100644 --- a/packages/tui/test/markdown.test.ts +++ b/packages/tui/test/markdown.test.ts @@ -1,6 +1,13 @@ import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test"; import { stripVTControlCharacters } from "node:util"; -import { clearRenderCache, Markdown, renderInlineMarkdown } from "@oh-my-pi/pi-tui/components/markdown"; +import { + autolinkSchemeScanIndex, + clearRenderCache, + Markdown, + mathStartIndex, + renderInlineMarkdown, + urlTokenPossible, +} from "@oh-my-pi/pi-tui/components/markdown"; import { setTerminalTextSizing, TERMINAL } from "@oh-my-pi/pi-tui/terminal-capabilities"; import { type Component, TUI } from "@oh-my-pi/pi-tui/tui"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; @@ -1940,9 +1947,11 @@ describe("Markdown.render reference stability", () => { }); it("does not share oversized renders through the L2 cache", () => { + // Fixture must exceed RENDER_CACHE_MAX_ENTRY_SIZE (256 KiB of rendered + // lines) so the entry is rejected and each render owns its array. const width = 80; const paragraph = `cache-budget sentinel ${"x".repeat(120)}`; - const largeText = Array.from({ length: 160 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n"); + const largeText = Array.from({ length: 1400 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n"); const first = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width); const second = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width); @@ -2319,3 +2328,146 @@ describe("Math rendering", () => { expect(lines[fxIdx + 1]).toContain("x > 0"); }); }); + +describe("inline start()/url-gate scanners (perf rewrites)", () => { + // The hand-rolled scanners replaced regex scans that marked runs on the + // remaining source at every inline position. They must return exactly what + // the old regexes returned for every input. + const OLD_MATH_START = /\$|\\\(|\\\[/; + const OLD_AUTOLINK_SCAN = /www\.|https?:\/\/|ftp:\/\//i; + // marked's bundled GFM inline url rule (verbatim, no flags). + const GFM_URL_REGEX = + /^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9-]+\.?)+[^\s<]*|^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/; + + const fixtures = [ + "", + "plain prose with no candidates at all", + "$x$ math first", + "prose then $inline$ math", + "prose then \\(paren\\) math", + "prose then \\[bracket\\] math", + "\\( before $ dollar", + "$ before \\( paren", + "backslash only \\ then ( apart", + "ends with backslash \\", + "ends with dollar $", + "www.example.com leading", + "see www.example.com mid-string", + "see WWW.EXAMPLE.COM upper", + "mixed WwW.case.com scan", + "http://example.com leading", + "prose http://example.com mid", + "prose HTTPS://EXAMPLE.COM upper", + "HtTpS://mixed.example", + "ftp://files.example mid ftp", + "prose FTP://FILES.EXAMPLE", + "ftps:// is not ftp:// until here ftp://x", + "wwww.overlap.example", + "hhttp://overlap.example", + "http:/ missing slash then https://real.example", + "www without dot www. with dot", + "w h f teaser chars but no scheme", + "user@example.com email", + "prose user.name+tag@example.co.uk", + "trailing at sign only@ ", + "@leading-at no local part", + "a".repeat(400), // long identifier run, no @ + `${"a".repeat(400)}@example.com`, // long local part (past gate scan limit) + "short@x", + "dots...and+plus_under-score@host.tld", + ]; + + it("mathStartIndex matches the old /\\$|\\\\\\(|\\\\\\[/ scan on every fixture", () => { + for (const src of fixtures) { + const m = OLD_MATH_START.exec(src); + expect(mathStartIndex(src)).toBe(m ? m.index : undefined); + } + }); + + it("autolinkSchemeScanIndex matches the old /www\\.|https?:\\/\\/|ftp:\\/\\//i scan on every fixture", () => { + for (const src of fixtures) { + const m = OLD_AUTOLINK_SCAN.exec(src); + expect(autolinkSchemeScanIndex(src)).toBe(m ? m.index : undefined); + } + }); + + it("urlTokenPossible is conservative: never false when the GFM url regex matches", () => { + for (const src of fixtures) { + if (GFM_URL_REGEX.test(src)) { + expect(urlTokenPossible(src)).toBeTrue(); + } + } + // And it actually gates: plain prose with no scheme/email head is rejected. + expect(urlTokenPossible("plain prose, nothing linkable here")).toBeFalse(); + expect(urlTokenPossible("@leading-at no local part")).toBeFalse(); + }); + + it("gated tokenizer still autolinks urls and emails end-to-end", () => { + const rendered = renderInlineMarkdown("see https://example.com and mail user@example.com now", { + ...defaultMarkdownTheme, + link: (text: string) => `${text}`, + }); + const plain = stripVTControlCharacters(rendered); + expect(plain).toContain("https://example.com"); + expect(plain).toContain("user@example.com"); + }); +}); + +describe("windowed lexing (documents past WINDOWED_LEX_MIN_BYTES)", () => { + // Large documents are lexed in bounded windows because Bun's regex engine + // rescans the whole remaining source for marked's `^`-anchored block rules. + // Every construct below straddles window cuts; a bad cut is visible in the + // rendered output. + afterEach(() => clearRenderCache()); + + const filler = (label: string, lines: number) => + Array.from({ length: lines }, (_, i) => `${label} paragraph ${i} with enough prose to fill a window.`).join( + "\n\n", + ); + + const plain = (text: string, width = 100) => + new Markdown(text, 0, 0, defaultMarkdownTheme) + .render(width) + .map(line => stripVTControlCharacters(line).trimEnd()); + + it("resolves a reference definition that lands in a later window", () => { + const doc = `Follow [the label][ref] first.\n\n${filler("body", 400)}\n\n[ref]: https://example.com/late\n`; + expect(doc.length).toBeGreaterThan(16 * 1024); + + const rendered = plain(doc, 120); + // The reflink resolved: marked emitted a link token (rendered as + // `label (href)`), so the raw `[label][ref]` syntax is gone and the + // definition line itself produced no output block of its own. + expect(rendered[0]).toBe("Follow the label (https://example.com/late) first."); + expect(rendered.filter(line => line.includes("https://example.com/late"))).toHaveLength(1); + }); + + it("keeps a fenced block longer than one window intact", () => { + const code = Array.from({ length: 200 }, (_, i) => `const value${i} = ${i};`).join("\n"); + const doc = `${filler("intro", 300)}\n\n\`\`\`ts\n${code}\n\`\`\`\n\n${filler("outro", 20)}`; + expect(code.length).toBeGreaterThan(2 * 1024); + + const rendered = plain(doc); + // Exactly one fence pair: a window cut inside the block would close and + // reopen it (or spill code lines into prose). + expect(rendered.filter(line => line.trimStart().startsWith("```"))).toHaveLength(2); + const first = rendered.findIndex(line => line.includes("const value0 = 0;")); + expect(first).toBeGreaterThan(-1); + for (let i = 0; i < 200; i++) { + expect(rendered[first + i]).toContain(`const value${i} = ${i};`); + } + }); + + it("numbers an ordered list continuously across window cuts", () => { + const items = Array.from({ length: 400 }, (_, i) => `${i + 1}. item ${i} padded with extra words to add bytes`); + const doc = `${filler("intro", 60)}\n\n${items.join("\n")}\n`; + expect(doc.length).toBeGreaterThan(16 * 1024); + + const rendered = plain(doc, 120); + for (const n of [1, 137, 400]) { + expect(rendered.some(line => line.includes(`${n}. item ${n - 1} `))).toBe(true); + } + // A window cut that restarted the list would renumber later items. + expect(rendered.filter(line => line.includes(" 1. item 0 ")).length).toBeLessThanOrEqual(1); + }); +});