perf(tui): optimized markdown lexing, editor input, and cursor updates

- Optimized markdown streaming lexing, tokenizers, and render cache limits.
- Improved editor input handling with bulk printable fast paths and iterative chunk processing.
- Added cursor visibility tracking and deduplication to prevent redundant terminal escape sequences.
- Added benchmarks and unit tests covering markdown streaming, incremental lexing, and editor fast paths.
This commit is contained in:
can1357
2026-07-27 20:33:39 +02:00
parent 6b42097d12
commit 1888b0999e
12 changed files with 1054 additions and 95 deletions
+10
View File
@@ -2,6 +2,16 @@
## [Unreleased]
### Added
- Added bulk-input fast path and iterative processing for bracketed paste in the editor
- Added windowed incremental lexing for large markdown documents
### Changed
- Optimized markdown URL tokenizer gate, inline math start scan, and autolink scheme scan for performance
- Deduplicated terminal cursor-visibility writes to skip redundant escape sequences
## [17.1.6] - 2026-07-27
### Fixed
+2 -2
View File
@@ -528,8 +528,8 @@ interface Terminal {
get columns(): number;
get rows(): number;
moveBy(lines: number): void;
hideCursor(): void;
showCursor(): void;
hideCursor(force?: boolean): void;
showCursor(force?: boolean): void;
clearLine(): void;
clearFromCursor(): void;
clearScreen(): void;
+92
View File
@@ -0,0 +1,92 @@
/**
* Streaming markdown render benchmark.
*
* Simulates a model streaming a long markdown message into one reused
* `Markdown` component (the interactive-mode hot path): the text grows in
* fixed-size deltas and the component re-renders after each delta.
*
* Exercises the profile hotspots from the 2026-07 capture: marked's GFM `url`
* tokenizer (73.3% self), inline extension `start()` scans, emStrong, and the
* streaming stable-prefix freeze (`#freezeStablePrefix`).
*
* Run: bun packages/tui/bench/markdown-stream.ts
*/
import { clearRenderCache, Markdown } from "../src/components/markdown";
import { defaultMarkdownTheme } from "../test/test-themes";
const WIDTH = 100;
const DELTA = 64; // chars revealed per streaming step
// --- Fixtures -------------------------------------------------------------
/** Long bullet list — the shape that defeats prefix freezing (list guard). */
function bulletList(items: number): string {
const lines: string[] = [];
for (let i = 0; i < items; i++) {
lines.push(
`- \`packages/tui/src/components/markdown_component_${i}.ts\` handles the ` +
`stable_prefix_freeze_path_${i} and re-lexes only the unfrozen tail, see ` +
`https://github.com/can1357/oh-my-pi/issues/${1000 + i} for details on token_${i}.`,
);
}
return `${lines.join("\n")}\n\n`;
}
/** Prose dense in email-branch pathology: long `[A-Za-z0-9._+-]+` runs with no `@`. */
function identifierProse(paragraphs: number): string {
const parts: string[] = [];
for (let i = 0; i < paragraphs; i++) {
parts.push(
`The resolver maps session_listing.scan_session_file.header_cache_v${i} onto ` +
`auth_broker.remote_store.filter_usage_reports_${i} while user${i}@example.com and ` +
`www.example${i}.org stay autolinked; identifiers like RENDER_CACHE_MAX_ENTRY_SIZE_${i} ` +
`and freeze.stable.prefix.tokens.v${i} must parse as plain *text* with **no** backtracking.`,
);
}
return `${parts.join("\n\n")}\n\n`;
}
function fences(count: number): string {
const parts: string[] = [];
for (let i = 0; i < count; i++) {
parts.push("```ts\nconst x_" + i + " = await fetch(\"https://api.example.com/v1/usage\");\n```\n");
}
return `${parts.join("\n")}\n`;
}
const DOC = identifierProse(20) + bulletList(120) + fences(8) + identifierProse(20) + bulletList(80);
// --- Bench ----------------------------------------------------------------
function streamOnce(text: string): number {
clearRenderCache();
const component = new Markdown("", 0, 0, defaultMarkdownTheme);
component.transientRenderCache = true;
const start = Bun.nanoseconds();
for (let len = DELTA; len < text.length; len += DELTA) {
component.setText(text.slice(0, len));
component.render(WIDTH);
}
component.setText(text);
component.render(WIDTH);
return (Bun.nanoseconds() - start) / 1e6;
}
function coldOnce(text: string): number {
clearRenderCache();
const start = Bun.nanoseconds();
new Markdown(text, 0, 0, defaultMarkdownTheme).render(WIDTH);
return (Bun.nanoseconds() - start) / 1e6;
}
console.log(`doc: ${DOC.length} chars, ${Math.ceil(DOC.length / DELTA)} streaming steps, width ${WIDTH}`);
// Warmup (JIT + regex compilation)
streamOnce(DOC.slice(0, 4096));
const cold = coldOnce(DOC);
console.log(`cold full render: ${cold.toFixed(1)}ms`);
const runs: number[] = [];
for (let i = 0; i < 3; i++) runs.push(streamOnce(DOC));
runs.sort((a, b) => a - b);
console.log(`streamed render (${runs.length} runs): min ${runs[0]!.toFixed(1)}ms, median ${runs[1]!.toFixed(1)}ms`);
+91 -42
View File
@@ -6,8 +6,8 @@ import {
midPromptSkillTokenMatches,
} from "../autocomplete";
import { BracketedPasteHandler, decodeReencodedPasteControls } from "../bracketed-paste";
import { getKeybindings, type KeybindingsManager } from "../keybindings";
import { extractPrintableText, matchesKey } from "../keys";
import { canonicalKeyId, getKeybindings, type KeybindingsManager } from "../keybindings";
import { extractPrintableText, matchesKey, parseKey } from "../keys";
import { KillRing } from "../kill-ring";
import type { SymbolTheme } from "../symbols";
import { type Component, CURSOR_MARKER, type Focusable } from "../tui";
@@ -341,6 +341,17 @@ function maxSegmentVisualCol(text: string, isLastSegment: boolean): number {
return isLastSegment ? total : Math.max(0, total - lastWidth);
}
/** True when every code unit is plain printable text: no C0 controls (so no
* ESC/CR/LF/TAB), no DEL, no C1 range — the same set `extractPrintableText`
* rejects. Such a run can never encode a key sequence. */
function isPlainTextRun(data: string): boolean {
for (let i = 0; i < data.length; i++) {
const code = data.charCodeAt(i);
if (code < 0x20 || code === 0x7f || (code >= 0x80 && code <= 0x9f)) return false;
}
return true;
}
const DEFAULT_PAGE_SCROLL_LINES = 10;
const MAX_UNDO_STACK = 100;
@@ -1126,12 +1137,30 @@ export class Editor implements Component, Focusable {
}
handleInput(data: string): void {
// Iterative, not recursive: the bytes trailing a completed bracketed
// paste (which may themselves contain further pastes) loop back here,
// so a fragmented paste stream can never grow the call stack.
let next: string | undefined = data;
while (next !== undefined && next.length > 0) {
next = this.#handleInputChunk(next);
}
}
/** Process one input chunk. Returns the unconsumed tail of a completed paste, if any. */
#handleInputChunk(data: string): string | undefined {
const kb = getKeybindings();
// Parse the sequence once; every binding probe below is then a set
// lookup instead of re-parsing `data` per probe (~35 probes per key).
const parsedKey = parseKey(data);
const canonical = parsedKey === undefined ? undefined : canonicalKeyId(parsedKey);
// Handle character jump mode (awaiting next character to jump to)
if (this.#jumpMode !== null) {
// Cancel if the hotkey is pressed again
if (kb.matches(data, "tui.editor.jumpForward") || kb.matches(data, "tui.editor.jumpBackward")) {
if (
kb.matchesCanonical(canonical, "tui.editor.jumpForward") ||
kb.matchesCanonical(canonical, "tui.editor.jumpBackward")
) {
this.#jumpMode = null;
return;
}
@@ -1154,12 +1183,23 @@ export class Editor implements Component, Focusable {
if (paste.pasteContent !== undefined) {
this.#handlePaste(paste.pasteContent);
if (paste.remaining.length > 0) {
this.handleInput(paste.remaining);
return paste.remaining;
}
}
return;
}
// Bulk printable fast path: a multi-scalar run of plain text (paste
// remainder, batched stdin) parses to no key, so no binding probe or
// special-key branch below can consume it — it always falls through to
// one #insertCharacter call. Take that path directly and skip the
// dispatch cascade. Runs containing ESC or control bytes (including
// \r/\n) keep the full path: those bytes carry key semantics.
if (canonical === undefined && data.length > 1 && isPlainTextRun(data)) {
this.#insertCharacter(data);
return;
}
// Handle special key combinations first
// Ctrl+C is reserved by parent components for app-level handling.
@@ -1170,7 +1210,7 @@ export class Editor implements Component, Focusable {
}
// Undo
if (kb.matches(data, "tui.editor.undo")) {
if (kb.matchesCanonical(canonical, "tui.editor.undo")) {
this.#applyUndo();
return;
}
@@ -1178,26 +1218,26 @@ export class Editor implements Component, Focusable {
// Handle autocomplete special keys first (but don't block other input)
if (this.#autocompleteState && this.#autocompleteList) {
// Escape - cancel autocomplete
if (kb.matches(data, "tui.select.cancel")) {
if (kb.matchesCanonical(canonical, "tui.select.cancel")) {
this.#cancelAutocomplete(true);
return;
}
// Let the autocomplete list handle navigation and selection
else if (
kb.matches(data, "tui.select.up") ||
kb.matches(data, "tui.select.down") ||
kb.matches(data, "tui.select.pageUp") ||
kb.matches(data, "tui.select.pageDown") ||
kb.matches(data, "tui.input.submit") ||
kb.matchesCanonical(canonical, "tui.select.up") ||
kb.matchesCanonical(canonical, "tui.select.down") ||
kb.matchesCanonical(canonical, "tui.select.pageUp") ||
kb.matchesCanonical(canonical, "tui.select.pageDown") ||
kb.matchesCanonical(canonical, "tui.input.submit") ||
data === "\n" ||
kb.matches(data, "tui.input.tab")
kb.matchesCanonical(canonical, "tui.input.tab")
) {
// Only pass navigation keys to the list, not Enter/Tab (we handle those directly)
if (
kb.matches(data, "tui.select.up") ||
kb.matches(data, "tui.select.down") ||
kb.matches(data, "tui.select.pageUp") ||
kb.matches(data, "tui.select.pageDown")
kb.matchesCanonical(canonical, "tui.select.up") ||
kb.matchesCanonical(canonical, "tui.select.down") ||
kb.matchesCanonical(canonical, "tui.select.pageUp") ||
kb.matchesCanonical(canonical, "tui.select.pageDown")
) {
this.#autocompleteList.handleInput(data);
this.onAutocompleteUpdate?.();
@@ -1205,7 +1245,7 @@ export class Editor implements Component, Focusable {
}
// If Tab was pressed, always apply the selection
if (kb.matches(data, "tui.input.tab")) {
if (kb.matchesCanonical(canonical, "tui.input.tab")) {
const selected = this.#autocompleteList.getSelectedItem();
// Check for stale autocomplete state due to buffer edits since last refresh
// (destructive keys or paste can outrun the debounced update).
@@ -1249,7 +1289,7 @@ export class Editor implements Component, Focusable {
// If Enter was pressed on a submitted slash command (not an absolute-path
// completion sharing the leading-slash prefix), apply and submit.
if (
(kb.matches(data, "tui.input.submit") || data === "\n") &&
(kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") &&
findLeadingSlashCommandStart(this.#autocompletePrefix) !== null &&
this.#isInSubmittedSlashCommandContext() &&
!this.#selectedCompletionIsPath()
@@ -1281,7 +1321,7 @@ export class Editor implements Component, Focusable {
// Don't return - fall through to submission logic
}
// Otherwise, apply the completion without submitting the surrounding draft.
else if (kb.matches(data, "tui.input.submit") || data === "\n") {
else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") {
const selected = this.#autocompleteList.getSelectedItem();
// Check for stale autocomplete state due to buffer edits since last refresh.
const currentLine = this.#state.lines[this.#state.cursorLine] ?? "";
@@ -1321,37 +1361,37 @@ export class Editor implements Component, Focusable {
}
// Tab key - context-aware completion (but not when already autocompleting)
if (kb.matches(data, "tui.input.tab") && !this.#autocompleteState) {
if (kb.matchesCanonical(canonical, "tui.input.tab") && !this.#autocompleteState) {
this.#handleTabCompletion();
return;
}
// Continue with rest of input handling
// Delete to end of line
if (kb.matches(data, "tui.editor.deleteToLineEnd")) {
if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineEnd")) {
this.#deleteToEndOfLine();
}
// Delete to start of line
else if (kb.matches(data, "tui.editor.deleteToLineStart")) {
else if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineStart")) {
this.#deleteToStartOfLine();
}
// Delete word backward. Registry defaults cover ctrl+w, alt+backspace,
// ctrl+backspace, and super+alt+backspace (Ghostty on macOS reports
// Option+Backspace as super+alt — kitty mod 11, see #2064).
else if (kb.matches(data, "tui.editor.deleteWordBackward")) {
else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordBackward")) {
this.#deleteWordBackwards();
}
// Delete word forward. Registry defaults cover alt+d/alt+delete and their
// super+alt variants for the same Ghostty quirk.
else if (kb.matches(data, "tui.editor.deleteWordForward")) {
else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordForward")) {
this.#deleteWordForwards();
}
// Yank from kill ring
else if (kb.matches(data, "tui.editor.yank")) {
else if (kb.matchesCanonical(canonical, "tui.editor.yank")) {
this.#yankFromKillRing();
}
// Yank-pop (cycle kill ring)
else if (kb.matches(data, "tui.editor.yankPop")) {
else if (kb.matchesCanonical(canonical, "tui.editor.yankPop")) {
this.#yankPop();
}
// Ctrl+A - Move to start of line
@@ -1376,7 +1416,7 @@ export class Editor implements Component, Focusable {
matchesKey(data, "ctrl+enter") || // Ctrl+Enter (Kitty/modifyOtherKeys, including lock bits/keypad Enter)
data === "\x1b\r" || // Option+Enter in some terminals (legacy)
data === "\x1b[13;2~" || // Shift+Enter in some terminals (legacy format)
kb.matches(data, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits)
kb.matchesCanonical(canonical, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits)
(data.length > 1 && data.includes("\x1b") && data.includes("\r")) ||
(data === "\n" && data.length === 1) // Shift+Enter from iTerm2 mapping
) {
@@ -1388,7 +1428,7 @@ export class Editor implements Component, Focusable {
this.#addNewLine();
}
// Plain Enter - submit (handles both legacy \r and Kitty protocol with lock bits)
else if (kb.matches(data, "tui.input.submit") || data === "\n") {
else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") {
// If submit is disabled, do nothing
if (this.disableSubmit) {
return;
@@ -1429,40 +1469,40 @@ export class Editor implements Component, Focusable {
this.#submitValue();
}
// Backspace (including Shift+Backspace)
else if (kb.matches(data, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) {
else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) {
this.#handleBackspace();
}
// Line navigation shortcuts (Home/End keys)
else if (kb.matches(data, "tui.editor.cursorLineStart")) {
else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineStart")) {
this.#moveToLineStart();
} else if (kb.matches(data, "tui.editor.cursorLineEnd")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineEnd")) {
this.#moveToLineEnd();
}
// Page navigation (PageUp/PageDown): page the editor viewport only. On a
// short draft this is a no-op — it never steps prompt history (that stays
// on Up/Down), so an idle empty editor swallows the keys instead of
// surprising the user by loading the previous prompt (#4754).
else if (kb.matches(data, "tui.editor.pageUp")) {
else if (kb.matchesCanonical(canonical, "tui.editor.pageUp")) {
this.#pageScroll(-1);
} else if (kb.matches(data, "tui.editor.pageDown")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.pageDown")) {
this.#pageScroll(1);
}
// Forward delete (Fn+Backspace or Delete key, including Shift+Delete)
else if (kb.matches(data, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) {
else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) {
this.#handleForwardDelete();
}
// Word navigation (Option/Alt + Arrow or Ctrl + Arrow)
else if (kb.matches(data, "tui.editor.cursorWordLeft")) {
else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordLeft")) {
// Word left
this.#resetKillSequence();
this.#moveWordBackwards();
} else if (kb.matches(data, "tui.editor.cursorWordRight")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordRight")) {
// Word right
this.#resetKillSequence();
this.#moveWordForwards();
}
// Arrow keys
else if (kb.matches(data, "tui.editor.cursorUp")) {
else if (kb.matchesCanonical(canonical, "tui.editor.cursorUp")) {
// Up - history navigation or cursor movement
if (this.#isEditorEmpty()) {
this.#navigateHistory(-1); // Start browsing history
@@ -1474,7 +1514,7 @@ export class Editor implements Component, Focusable {
} else {
this.#moveCursor(-1, 0); // Cursor movement (within text or history entry)
}
} else if (kb.matches(data, "tui.editor.cursorDown")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorDown")) {
// Down - history navigation or cursor movement
if (this.#historyIndex > -1 && this.#isOnLastVisualLine()) {
this.#navigateHistory(1); // Navigate to newer history entry or clear
@@ -1484,10 +1524,10 @@ export class Editor implements Component, Focusable {
} else {
this.#moveCursor(1, 0); // Cursor movement (within text or history entry)
}
} else if (kb.matches(data, "tui.editor.cursorRight")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorRight")) {
// Right
this.#moveCursor(0, 1);
} else if (kb.matches(data, "tui.editor.cursorLeft")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorLeft")) {
// Left
this.#moveCursor(0, -1);
}
@@ -1496,9 +1536,9 @@ export class Editor implements Component, Focusable {
this.#insertCharacter(" ");
}
// Character jump mode triggers
else if (kb.matches(data, "tui.editor.jumpForward")) {
else if (kb.matchesCanonical(canonical, "tui.editor.jumpForward")) {
this.#jumpMode = "forward";
} else if (kb.matches(data, "tui.editor.jumpBackward")) {
} else if (kb.matchesCanonical(canonical, "tui.editor.jumpBackward")) {
this.#jumpMode = "backward";
}
// Printable keystrokes, including Kitty CSI-u text-producing sequences.
@@ -1630,6 +1670,15 @@ export class Editor implements Component, Focusable {
return this.#state.lines.join("\n");
}
/** Whether the buffer text equals `value`, without `getText()`'s full join —
* O(1) for the hot per-keystroke probes against short single-line values. */
textEquals(value: string): boolean {
const lines = this.#state.lines;
if (lines.length === 1) return lines[0] === value;
if (value.indexOf("\n") === -1) return false;
return this.getText() === value;
}
#expandPasteMarkers(text: string): string {
let result = text;
for (const [pasteId, pasteContent] of this.#pastes) {
+310 -40
View File
@@ -1,5 +1,5 @@
import { LRUCache } from "lru-cache/raw";
import { Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked";
import { Lexer, Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked";
import { latexToBlock } from "../latex-block";
import { inlineMathSpanEnd, isBareMathEnvironment, latexToUnicode } from "../latex-to-unicode";
import type { SymbolTheme } from "../symbols";
@@ -492,12 +492,25 @@ const customHrExtension: TokenizerAndRendererExtension = {
},
};
// Leftmost-match scan replacing /\$|\\\(|\\\[/ in mathExtension.start —
// marked calls start() on the remaining source at every inline position, so
// the regex alternation showed up in CPU profiles (part of a ~4.3% start()
// tail). Three indexOf scans yield the identical leftmost index.
/** @internal exported for tests — must stay index-identical to the old regex scan. */
export function mathStartIndex(src: string): number | undefined {
let best = src.indexOf("$");
const paren = src.indexOf("\\(");
if (paren !== -1 && (best === -1 || paren < best)) best = paren;
const bracket = src.indexOf("\\[");
if (bracket !== -1 && (best === -1 || bracket < best)) best = bracket;
return best === -1 ? undefined : best;
}
const mathExtension: TokenizerAndRendererExtension = {
name: "math",
level: "inline",
start(src) {
const m = /\$|\\\(|\\\[/.exec(src);
return m ? m.index : undefined;
return mathStartIndex(src);
},
tokenizer(src) {
if (src.startsWith("$$")) {
@@ -614,14 +627,63 @@ const mathEnvBlockExtension: TokenizerAndRendererExtension = {
// tokenizer at a valid start. Candidates at a legal boundary fall through
// (return undefined) to marked's own autolink handling unchanged.
const AUTOLINK_SCHEME_REGEX = /^(?:www\.|https?:\/\/|ftp:\/\/)/i;
const AUTOLINK_SCHEME_SCAN = /www\.|https?:\/\/|ftp:\/\//i;
// Case-insensitive scheme scan replacing /www\.|https?:\/\/|ftp:\/\//i in
// boundedAutolinkExtension.start — like mathStartIndex above, this runs on the
// remaining source at every inline position (part of a ~4.3% CPU start() scan
// tail in profiles). charCode-only: no allocation, no toLowerCase copies.
// `| 32` lower-cases ASCII letters; `.`/`:`/`/` are compared exactly, matching
// the regex's ASCII-only `i` semantics. charCodeAt past the end returns NaN,
// which fails every comparison, so no explicit bounds checks are needed.
function isAutolinkSchemeAt(src: string, i: number): boolean {
const c = src.charCodeAt(i) | 32;
if (c === 119 /* w */) {
// www.
return (
(src.charCodeAt(i + 1) | 32) === 119 &&
(src.charCodeAt(i + 2) | 32) === 119 &&
src.charCodeAt(i + 3) === 46 /* . */
);
}
if (c === 104 /* h */) {
// http:// | https://
if (
(src.charCodeAt(i + 1) | 32) !== 116 /* t */ ||
(src.charCodeAt(i + 2) | 32) !== 116 /* t */ ||
(src.charCodeAt(i + 3) | 32) !== 112 /* p */
) {
return false;
}
let j = i + 4;
if ((src.charCodeAt(j) | 32) === 115 /* s */) j++;
return src.charCodeAt(j) === 58 /* : */ && src.charCodeAt(j + 1) === 47 /* / */ && src.charCodeAt(j + 2) === 47;
}
if (c === 102 /* f */) {
// ftp://
return (
(src.charCodeAt(i + 1) | 32) === 116 /* t */ &&
(src.charCodeAt(i + 2) | 32) === 112 /* p */ &&
src.charCodeAt(i + 3) === 58 /* : */ &&
src.charCodeAt(i + 4) === 47 /* / */ &&
src.charCodeAt(i + 5) === 47 /* / */
);
}
return false;
}
/** @internal exported for tests — must stay index-identical to the old regex scan. */
export function autolinkSchemeScanIndex(src: string): number | undefined {
for (let i = 0; i < src.length; i++) {
const c = src.charCodeAt(i) | 32;
if ((c === 119 || c === 104 || c === 102) && isAutolinkSchemeAt(src, i)) return i;
}
return undefined;
}
const VALID_AUTOLINK_LEFT_BOUNDARY = /[\s*_~(]/;
const boundedAutolinkExtension: TokenizerAndRendererExtension = {
name: "boundedAutolink",
level: "inline",
start(src) {
const m = AUTOLINK_SCHEME_SCAN.exec(src);
return m ? m.index : undefined;
return autolinkSchemeScanIndex(src);
},
tokenizer(src, tokens) {
const match = AUTOLINK_SCHEME_REGEX.exec(src);
@@ -639,6 +701,66 @@ markdownParser.use({
extensions: [customHrExtension, mathBlockExtension, mathEnvBlockExtension, mathExtension, boundedAutolinkExtension],
});
// ---------------------------------------------------------------------------
// GFM `url` tokenizer gate
// ---------------------------------------------------------------------------
// marked tries the bundled GFM `url` tokenizer at every inline tokenization
// step, and its regex is expensive to FAIL: the email alternative
// `^[A-Za-z0-9._+-]+(@)…` linearly consumes an identifier run, then backtracks
// it one character at a time when no `@` follows. A 71414-sample / 1ms CPU
// profile of the TUI put 73.3% of total CPU (74.9s of a 102s capture) inside
// this single regex. The override below runs an O(bounded) charCode gate first
// and only falls through to the built-in tokenizer — by returning `false`,
// marked's tokenizer-override fallback contract — when a match is possible.
//
// Conservativeness argument. The built-in rule (no flags) is
// /^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.?)+[^\s<]*
// |^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/
// Both alternatives are anchored, so any match constrains the head of src:
// • Branch 1 requires src to start with `http://`, `https://`, `ftp://`
// (scheme letters in any case) or lowercase `www.`. The gate accepts all of
// these via isAutolinkSchemeAt(src, 0); it also over-accepts `WWW.`, a
// harmless false positive (the built-in regex simply fails to match).
// • Branch 2 requires src to start with one-or-more chars from
// `[A-Za-z0-9._+-]` immediately followed by `@`. The gate scans that exact
// class: if the run ends within URL_GATE_EMAIL_SCAN_LIMIT chars it accepts
// iff the terminator is `@`; a run reaching the limit is accepted
// unconditionally. Every src branch 2 can match is therefore accepted —
// the gate never rejects a src the built-in regex would match.
const URL_GATE_EMAIL_SCAN_LIMIT = 320;
/** @internal exported for tests — must never return false for a src the built-in url regex matches. */
export function urlTokenPossible(src: string): boolean {
if (isAutolinkSchemeAt(src, 0)) return true;
let i = 0;
while (i < URL_GATE_EMAIL_SCAN_LIMIT) {
const c = src.charCodeAt(i);
const isLocalChar =
(c >= 97 && c <= 122) /* a-z */ ||
(c >= 65 && c <= 90) /* A-Z */ ||
(c >= 48 && c <= 57) /* 0-9 */ ||
c === 46 /* . */ ||
c === 95 /* _ */ ||
c === 43 /* + */ ||
c === 45; /* - */
if (!isLocalChar) break;
i++;
}
if (i === 0) return false;
if (i >= URL_GATE_EMAIL_SCAN_LIMIT) return true; // over-long run: give up conservatively
return src.charCodeAt(i) === 64 /* @ */;
}
markdownParser.use({
tokenizer: {
url(src: string): Tokens.Link | undefined | false {
// `false` → marked falls back to the built-in `url` tokenizer;
// `undefined` → no url token here, built-in never runs.
return urlTokenPossible(src) ? false : undefined;
},
},
});
// ---------------------------------------------------------------------------
// Module-level LRU render cache
// ---------------------------------------------------------------------------
@@ -649,8 +771,8 @@ markdownParser.use({
// (Rust FFI) work for content/layout combinations already seen this session.
const RENDER_CACHE_MAX = 256; // sane cap: ~256 distinct message × width combos
const RENDER_CACHE_MAX_SIZE = 512 * 1024;
const RENDER_CACHE_MAX_ENTRY_SIZE = 32 * 1024;
const RENDER_CACHE_MAX_SIZE = 4 * 1024 * 1024;
const RENDER_CACHE_MAX_ENTRY_SIZE = 256 * 1024;
const EMPTY_RENDER_LINES: readonly string[] = [];
interface RenderCacheEntry {
@@ -684,6 +806,179 @@ function renderCacheEntrySize(entry: RenderCacheEntry): number {
// over-matching is safe (it only costs the fast path), under-matching is not.
const HAS_REF_DEF = /^ {0,3}\[(?:\\.|[^\]\\])+\]:/m;
// marked's list tokenizer (Tokenizer.list, marked v18) continues a list across
// blank lines only when the remaining source matches
// `listItemRegex(marker)` = `^( {0,3}${marker})((?:[\t ][^\n]*)?(?:\n|$))`,
// where `marker` is the exact bullet char for unordered lists (`\${char}`) or
// 1-9 digits plus the exact delimiter for ordered lists (`\d{1,9}\${delim}`).
// The marker is derived from the list's FIRST item (`n = t[1].trim()`), which
// sits at the start of a top-level list token's raw:
const LIST_MARKER_RE = /^ {0,3}(?:([*+-])|\d{1,9}([.)]))/;
// Streaming-freeze equivalence invariant: lex(prefix) ++ lex(tail) must equal
// lex(full text) — for the CURRENT text and for every append-only extension of
// it, because a frozen prefix is sticky (it keeps being reused while the text
// grows). At a blank-line (`\n\n`) cut directly after a top-level `list`
// token, the only construct that can straddle the cut is a continuation item
// of that list: marked consumed the blank line into the last item's raw and
// re-ran `listItemRegex` at exactly `tailStart`, merging a same-marker item
// into one renumbered loose list. The cut is safe only when that regex can
// NEVER match at `tailStart`, no matter what is appended later.
//
// Append-only growth means existing characters are immutable while new ones
// may appear after them, so "closed" may only be concluded from a present
// character that contradicts every possible continuation (e.g. tail "1x" can
// never grow into an ordered item, but tail "1" can become "1. c"). Running
// out of text mid-marker therefore answers "may continue".
//
// Returns true when the tail could still continue the list (or the list's
// marker is unrecognizable) — the conservative "don't freeze" answer. marked
// may break the list anyway when the matching line is also an hr (`- - -`);
// treating that as "may continue" merely skips a freeze, never corrupts one.
function listMayContinueAt(text: string, tailStart: number, listRaw: string): boolean {
const marker = LIST_MARKER_RE.exec(listRaw);
if (marker === null) return true; // unrecognized list shape — stay conservative
const n = text.length;
let i = tailStart;
// `listItemRegex` allows up to 3 leading spaces (the caller's next-char
// guard rejects whitespace at the final cut, but mirror the rule exactly).
while (i < n && i - tailStart < 3 && text.charCodeAt(i) === 0x20 /* space */) i++;
if (i >= n) return true;
const bullet = marker[1];
if (bullet !== undefined) {
if (text[i] !== bullet) return false; // wrong marker char — closed forever
i++;
} else {
// Ordered: 1-9 digits, then the same `.`/`)` delimiter.
let digits = 0;
while (i < n && digits < 10) {
const c = text.charCodeAt(i);
if (c < 0x30 /* 0 */ || c > 0x39 /* 9 */) break;
digits++;
i++;
}
if (digits === 0 || digits > 9) return false; // no digit run / too long — closed forever
if (i >= n) return true; // delimiter (or more digits) may still arrive
if (text[i] !== marker[2]) return false; // wrong delimiter — closed forever
i++;
}
// After the marker: `(?:[\t ][^\n]*)?(?:\n|$)` — tab/space + anything, a
// bare newline, or end-of-input (which appends can still extend).
if (i >= n) return true;
const after = text.charCodeAt(i);
return after === 0x20 /* space */ || after === 0x09 /* tab */ || after === 0x0a /* \n */;
}
const NO_BLOCK_BOUNDARY = { end: 0, count: 0 } as const;
/**
* Offset just past the last token in `tokens` that closes a block on a hard
* `"\n\n"` break, together with the number of tokens up to and including it.
* `count === 0` means the run holds no usable boundary.
*
* `base` is where `tokens[0]` starts inside `text`. A boundary qualifies only
* when splitting there is invisible to the lexer, i.e. `lex(head) ++ lex(tail)
* === lex(text)`:
* - The break must sit inside `text`. At end-of-text the next character is
* unknown (and, while streaming, may still arrive), so the cut is deferred.
* - The next character must start real block content. Whitespace means the
* block separator straddles the cut — e.g. a fence followed by
* `"\n\n\n- list"` — and the two lexes desync.
* - A preceding `list` must be provably closed: CommonMark lets a same-marker
* item continue the list across the blank line, and marked merges both into
* one renumbered loose list (`listMayContinueAt`).
*/
function stableBlockBoundary(text: string, base: number, tokens: Token[]): { end: number; count: number } {
let pos = base;
let end = 0;
let count = 0;
for (let i = 0; i < tokens.length; i++) {
const raw = tokens[i].raw;
const tokenEnd = pos + raw.length;
if (raw.endsWith("\n\n")) {
const prev = i > 0 ? tokens[i - 1] : undefined;
if (prev === undefined || prev.type !== "list" || !listMayContinueAt(text, tokenEnd, prev.raw)) {
end = tokenEnd;
count = i + 1;
}
}
pos = tokenEnd;
}
if (count === 0 || end >= text.length) return NO_BLOCK_BOUNDARY;
const next = text.charCodeAt(end);
if (next === 0x20 /* space */ || next === 0x0a /* \n */) return NO_BLOCK_BOUNDARY;
return { end, count };
}
// Bun's regex engine skips the start-anchor optimization for several of marked's
// block rules — `hr`, `lheading`, `table` and `html` are `^`-anchored
// alternations of quantified branches — so each failing `exec` rescans the whole
// remaining source instead of stopping at offset 0. Lexing is then quadratic in
// document length: an 800 KB message costs ~41 s under Bun where Node/V8 needs
// ~60 ms, and it runs on the render path, freezing the UI. Bounded windows keep
// every scan short and restore linear behavior (~0.7 s for that same message).
const LEX_WINDOW_BYTES = 2 * 1024;
// Under this size a single pass beats probing for window boundaries; the
// crossover measured on pathological Markdown sits around 16 KB.
const WINDOWED_LEX_MIN_BYTES = 16 * 1024;
/**
* Lex `text` in bounded windows, producing the exact token stream
* `markdownParser.lexer(text)` would.
*
* Window cuts come from marked itself: a throwaway BLOCK-ONLY probe lex of the
* window reports its last stable block boundary ({@link stableBlockBoundary})
* and only that confirmed segment is handed to the real lexer; a window
* holding no boundary doubles until it finds one or reaches the end. Probes
* never run inline tokenization (their inlineQueue is discarded) — a boundary
* is a property of block structure alone, and probe inline passes were the
* dominant cost of an earlier revision. Block tokenization runs per window
* while inline tokenization is deferred to the end — mirroring `Lexer.lex` —
* so a `[label]: dest` definition anywhere in the document still resolves for
* every inline span.
*
* A boundary requires some top-level token whose raw ends in `"\n\n"`, so a
* window that contains no blank line cannot cut: each round starts at the next
* `"\n\n"` (skipping straight to the end when there is none — e.g. a tail
* that is one long tight list) instead of probing sizes that cannot succeed.
*/
function lexWindowed(text: string): Token[] {
const lexer = new Lexer(markdownParser.defaults);
let offset = 0;
while (offset < text.length) {
let segment = "";
const nextBlank = text.indexOf("\n\n", offset);
if (nextBlank === -1) {
segment = text.slice(offset);
} else {
const minSize = Math.max(LEX_WINDOW_BYTES, nextBlank + 2 - offset);
for (let size = minSize; segment.length === 0; size *= 2) {
if (offset + size >= text.length) {
segment = text.slice(offset);
break;
}
const probe = new Lexer(markdownParser.defaults);
probe.blockTokens(text.slice(offset, offset + size), probe.tokens);
const boundary = stableBlockBoundary(text, offset, probe.tokens);
if (boundary.count > 0) segment = text.slice(offset, boundary.end);
}
}
lexer.blockTokens(segment, lexer.tokens);
offset += segment.length;
}
for (const queued of lexer.inlineQueue) lexer.inlineTokens(queued.src, queued.tokens);
lexer.inlineQueue = [];
return lexer.tokens;
}
/** Lex a whole document, windowing anything large enough for the quadratic scan to bite. */
function lexDocument(text: string): Token[] {
// A CR shifts every `raw` span (marked normalizes CRLF before tokenizing), so
// window offsets would address the wrong characters — lex those in one pass.
if (text.length < WINDOWED_LEX_MIN_BYTES || text.includes("\r")) return markdownParser.lexer(text);
return lexWindowed(text);
}
/** Drop all L2 cache entries. Call on theme change to prevent stale styled output. */
export function clearRenderCache(): void {
renderCache.clear();
@@ -1219,12 +1514,12 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ
text.length > prefix.length &&
text.startsWith(prefix)
) {
const tailTokens = markdownParser.lexer(text.slice(prefix.length));
const tailTokens = lexDocument(text.slice(prefix.length));
const tokens = [...prefixTokens, ...tailTokens];
this.#freezeStablePrefix(text, tokens, { preserveExisting: true });
return tokens;
}
const tokens = markdownParser.lexer(text);
const tokens = lexDocument(text);
if (canStream) {
this.#freezeStablePrefix(text, tokens, { preserveExisting: false });
} else {
@@ -1241,36 +1536,11 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ
// reference definitions, so each token's `raw` is a verbatim slice of `text`
// and the summed offsets address `text` exactly.
#freezeStablePrefix(text: string, tokens: Token[], opts: { preserveExisting: boolean }): void {
let pos = 0;
let frozenEnd = 0;
let frozenCount = 0;
for (let i = 0; i < tokens.length; i++) {
const raw = tokens[i].raw;
const end = pos + raw.length;
// A `space` token ending in "\n\n" closes the preceding block, but a
// `list` before it can still be extended by a following same-marker
// item across the blank line (CommonMark loose-list continuation),
// which marked merges into one renumbered loose list. Freezing across
// such a cut would keep the lists separate. Never freeze right after a
// list — it stays in the re-lexed tail.
if (raw.endsWith("\n\n") && tokens[i - 1]?.type !== "list") {
frozenEnd = end;
frozenCount = i + 1;
}
pos = end;
}
// Freeze only when the tail begins with real block content. If the next
// char is whitespace (an extra blank line, or an indented continuation),
// the block separator straddles the cut and lex(prefix)++lex(tail) would
// desync from a full lex — e.g. a fence followed by "\n\n\n- list". When
// frozenEnd is at end-of-text the next char is unknown, so defer.
if (frozenCount > 0 && frozenEnd < text.length) {
const next = text.charCodeAt(frozenEnd);
if (next !== 0x20 /* space */ && next !== 0x0a /* \n */) {
this.#streamPrefixText = text.slice(0, frozenEnd);
this.#streamPrefixTokens = tokens.slice(0, frozenCount);
return;
}
const frozen = stableBlockBoundary(text, 0, tokens);
if (frozen.count > 0) {
this.#streamPrefixText = text.slice(0, frozen.end);
this.#streamPrefixTokens = tokens.slice(0, frozen.count);
return;
}
if (!opts.preserveExisting) {
+11 -2
View File
@@ -288,8 +288,17 @@ export class KeybindingsManager {
matches(data: string, keybinding: Keybinding): boolean {
const parsed = parseKey(data);
if (parsed === undefined) return false;
const matchKeys = this.#matchKeysById.get(keybinding);
return matchKeys?.has(canonicalKeyId(parsed)) ?? false;
return this.matchesCanonical(canonicalKeyId(parsed), keybinding);
}
/**
* Set-lookup variant of {@link matches} for hot input paths: the caller
* parses `data` once (`parseKey` + `canonicalKeyId`) and probes many
* bindings without re-parsing the raw sequence per probe.
*/
matchesCanonical(canonical: string | undefined, keybinding: Keybinding): boolean {
if (canonical === undefined) return false;
return this.#matchKeysById.get(keybinding)?.has(canonical) ?? false;
}
getKeys(keybinding: Keybinding): KeyId[] {
+46 -6
View File
@@ -286,7 +286,7 @@ export function emergencyTerminalRestore(): void {
terminal.write("\x1b[?1049l");
altScreenActive = false;
}
terminal.showCursor();
terminal.showCursor(true);
} else if (terminalEverStarted && !isTerminalHeadless()) {
// Blind restore only if we know a terminal was started but lost track of it
// This avoids writing escape sequences for non-TUI commands (grep, commit, etc.)
@@ -362,9 +362,11 @@ export interface Terminal {
// Cursor positioning (relative to current position)
moveBy(lines: number): void; // Move cursor up (negative) or down (positive) by N lines
// Cursor visibility
hideCursor(): void; // Hide the cursor
showCursor(): void; // Show the cursor
// Cursor visibility. Same-state calls are deduped against the visibility
// last written to the terminal; pass force=true to write unconditionally
// (crash/exit restore paths).
hideCursor(force?: boolean): void; // Hide the cursor
showCursor(force?: boolean): void; // Show the cursor
// Clear operations
clearLine(): void; // Clear current line
@@ -478,6 +480,12 @@ export class ProcessTerminal implements Terminal {
this.#markTerminalDisconnected("stdin failed", err);
};
#dead = false;
// Last cursor visibility written to the terminal, sniffed from every
// outgoing sequence (frame buffers embed their own ?25h/?25l), so
// hideCursor()/showCursor() can skip same-state writes. `undefined` =
// unknown (fresh start, resize, or an alt-screen switch newer than the
// last cursor sequence — some hosts keep DECTCEM per buffer).
#cursorVisible: boolean | undefined;
// Captured at construction and re-read at start(): when true, every real
// terminal side effect (writes, probes, raw mode, SIGWINCH, timers) is
// suppressed. Defaults on under `bun test` — see isTerminalHeadless().
@@ -575,6 +583,8 @@ export class ProcessTerminal implements Terminal {
this.#inputHandler = onInput;
this.#resizeHandler = onResize;
this.#disconnectHandler = onDisconnect;
// The host terminal's cursor visibility is unknown until we write it.
this.#cursorVisible = undefined;
// Headless (tests): suppress every real-terminal side effect. Skip raw
// mode, stdin listeners, capability probes, SIGWINCH, and emergency-restore
@@ -620,6 +630,9 @@ export class ProcessTerminal implements Terminal {
// dimensions before firing `resize`, so it is authoritative for geometry:
// reconcile any stale cached DEC 2048 report before notifying the renderer.
this.#stdoutResizeListener = () => {
// Conservative: some hosts reset modes across a resize/reattach, so
// re-establish cursor visibility on the next explicit call.
this.#cursorVisible = undefined;
this.#reconcileInBandGeometryOnResize();
this.#resizeHandler?.();
};
@@ -1474,6 +1487,9 @@ export class ProcessTerminal implements Terminal {
}
this.#stdoutErrorCleanup?.();
this.#stdoutErrorCleanup = undefined;
// After stop() the terminal is shared with other writers; visibility
// tracking is only meaningful while this instance owns the TTY.
this.#cursorVisible = undefined;
}
#ensureStdoutErrorHandler(): void {
@@ -1527,6 +1543,7 @@ export class ProcessTerminal implements Terminal {
// files). They serve no purpose there and would surface as visible noise.
if (!process.stdout.isTTY) return;
this.#ensureStdoutErrorHandler();
this.#trackCursorVisibility(data);
// A console-sharing child process may have flipped the console codepage
// away from UTF-8; repair it before any bytes hit WriteFile so no frame
// is ever translated through an OEM codepage. See ensureWindowsConsoleUtf8.
@@ -1578,14 +1595,37 @@ export class ProcessTerminal implements Terminal {
// lines === 0: no movement
}
hideCursor(): void {
hideCursor(force = false): void {
if (!force && this.#cursorVisible === false) return;
this.#safeWrite("\x1b[?25l");
}
showCursor(): void {
showCursor(force = false): void {
if (!force && this.#cursorVisible === true) return;
this.#safeWrite("\x1b[?25h");
}
/**
* Sniff outgoing data for the last cursor-visibility change so the tracked
* state stays correct for sequences embedded in frame buffers
* (TUI#cursorControlSequence appends ?25h/?25l inside the paint write). An
* alt-screen switch (DECSET/DECRST 1049) newer than the last cursor
* sequence resets tracking to unknown: some hosts keep DECTCEM per buffer.
*/
#trackCursorVisibility(data: string): void {
let idx = data.lastIndexOf("\x1b[?25");
while (idx !== -1) {
const final = data.charCodeAt(idx + 5);
if (final === 0x68 /* h */ || final === 0x6c /* l */) break;
idx = idx === 0 ? -1 : data.lastIndexOf("\x1b[?25", idx - 1);
}
if (data.lastIndexOf("\x1b[?1049") > idx) {
this.#cursorVisible = undefined;
return;
}
if (idx !== -1) this.#cursorVisible = data.charCodeAt(idx + 5) === 0x68;
}
clearLine(): void {
this.#safeWrite("\x1b[K");
}
+3 -1
View File
@@ -1875,7 +1875,9 @@ export class TUI extends Container {
this.terminal.write(targetRow <= viewportBottom ? "\r" : "\r\n");
}
this.terminal.showCursor();
// Force: the parent shell needs the cursor back regardless of what the
// terminal-level dedupe believes was last written.
this.terminal.showCursor(true);
this.#forgetHardwareCursorState();
this.terminal.stop();
}
@@ -0,0 +1,148 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { ProcessTerminal } from "@oh-my-pi/pi-tui/terminal";
import { setTerminalHeadless } from "@oh-my-pi/pi-utils";
// ProcessTerminal dedupes cursor-visibility writes: hideCursor()/showCursor()
// skip the ?25l/?25h escape when the terminal already holds that state. The
// tracked state is sniffed from every outgoing write, so cursor sequences
// embedded in frame buffers (TUI appends ?25h/?25l inside the paint write)
// keep it in sync, and an alt-screen switch resets it to unknown. Crash/exit
// restore paths pass force=true and must always write.
const HIDE = "\x1b[?25l";
const SHOW = "\x1b[?25h";
const stdinIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "isTTY");
const stdoutIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "isTTY");
const stdinSetRawModeDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "setRawMode");
let previousHeadless = false;
function restoreProperty(target: object, key: string, descriptor: PropertyDescriptor | undefined): void {
if (descriptor) {
Object.defineProperty(target, key, descriptor);
return;
}
delete (target as Record<string, unknown>)[key];
}
function startCapturedTerminal() {
const writes: string[] = [];
Object.defineProperty(process.stdin, "isTTY", { value: true, configurable: true });
Object.defineProperty(process.stdout, "isTTY", { value: true, configurable: true });
Object.defineProperty(process.stdin, "setRawMode", { value: vi.fn(), configurable: true });
vi.spyOn(process, "kill").mockReturnValue(true);
vi.spyOn(process.stdin, "resume").mockImplementation(() => process.stdin);
vi.spyOn(process.stdin, "pause").mockImplementation(() => process.stdin);
vi.spyOn(process.stdin, "setEncoding").mockImplementation(() => process.stdin);
vi.spyOn(process.stdout, "write").mockImplementation(chunk => {
writes.push(String(chunk));
return true;
});
const terminal = new ProcessTerminal();
terminal.start(
() => {},
() => {},
);
writes.length = 0;
return { terminal, writes };
}
describe("ProcessTerminal cursor-visibility dedupe", () => {
beforeEach(() => {
previousHeadless = setTerminalHeadless(false);
});
afterEach(() => {
setTerminalHeadless(previousHeadless);
vi.restoreAllMocks();
restoreProperty(process.stdin, "isTTY", stdinIsTtyDescriptor);
restoreProperty(process.stdout, "isTTY", stdoutIsTtyDescriptor);
restoreProperty(process.stdin, "setRawMode", stdinSetRawModeDescriptor);
});
it("writes each visibility change once and skips same-state repeats", () => {
const { terminal, writes } = startCapturedTerminal();
terminal.hideCursor();
terminal.hideCursor();
terminal.hideCursor();
expect(writes).toEqual([HIDE]);
terminal.showCursor();
terminal.showCursor();
expect(writes).toEqual([HIDE, SHOW]);
terminal.hideCursor();
expect(writes).toEqual([HIDE, SHOW, HIDE]);
terminal.stop();
});
it("tracks cursor sequences embedded in frame writes", () => {
const { terminal, writes } = startCapturedTerminal();
// A paint that repositions the hardware cursor ends by showing it.
terminal.write(`\x1b[2Bframe content\x1b[5G${SHOW}\x1b[?2026l`);
writes.length = 0;
terminal.showCursor(); // already visible per the frame write
expect(writes).toEqual([]);
terminal.hideCursor(); // state change: must write
expect(writes).toEqual([HIDE]);
terminal.stop();
});
it("honors the last of multiple cursor sequences in one write", () => {
const { terminal, writes } = startCapturedTerminal();
terminal.write(`${SHOW}overlay paint${HIDE}`);
writes.length = 0;
terminal.hideCursor();
expect(writes).toEqual([]);
terminal.showCursor();
expect(writes).toEqual([SHOW]);
terminal.stop();
});
it("force-writes regardless of tracked state (crash/exit restore contract)", () => {
const { terminal, writes } = startCapturedTerminal();
terminal.showCursor();
terminal.showCursor(true);
terminal.showCursor(true);
expect(writes).toEqual([SHOW, SHOW, SHOW]);
terminal.hideCursor();
terminal.hideCursor(true);
expect(writes).toEqual([SHOW, SHOW, SHOW, HIDE, HIDE]);
terminal.stop();
});
it("resets tracking to unknown when an alt-screen switch follows the last cursor sequence", () => {
const { terminal, writes } = startCapturedTerminal();
terminal.hideCursor();
// Alt-screen enter after the hide: some hosts keep DECTCEM per buffer,
// so the tracked state is no longer trustworthy.
terminal.write("\x1b[?1049h\x1b[2J");
writes.length = 0;
terminal.hideCursor();
expect(writes).toEqual([HIDE]);
terminal.stop();
});
it("does not confuse other private modes with cursor visibility", () => {
const { terminal, writes } = startCapturedTerminal();
terminal.hideCursor();
writes.length = 0;
// Neither DECRQM on mode 25 nor unrelated ?25xx modes change visibility.
terminal.write("\x1b[?25$p\x1b[?2026h");
terminal.hideCursor();
expect(writes).toEqual(["\x1b[?25$p\x1b[?2026h"]);
terminal.stop();
});
});
+104
View File
@@ -2396,6 +2396,110 @@ describe("Editor component", () => {
});
});
describe("Bulk input fast path and paste iteration", () => {
it("produces identical state for a chunked paste and a single-sequence paste", () => {
const content = "alpha beta\ngamma delta\nepsilon";
const single = new Editor(defaultEditorTheme);
single.handleInput(`\x1b[200~${content}\x1b[201~`);
const chunked = new Editor(defaultEditorTheme);
chunked.handleInput("\x1b[200~");
for (const ch of content) chunked.handleInput(ch);
chunked.handleInput("\x1b[201~");
expect(chunked.getText()).toBe(single.getText());
expect(chunked.getCursor()).toEqual(single.getCursor());
});
it("normalizes CRLF identically for single and chunked paste delivery", () => {
const single = new Editor(defaultEditorTheme);
single.handleInput("\x1b[200~one\r\ntwo\rthree\x1b[201~");
const chunked = new Editor(defaultEditorTheme);
chunked.handleInput("\x1b[200~one\r");
chunked.handleInput("\ntwo");
chunked.handleInput("\rthree\x1b[201~");
expect(single.getText()).toBe("one\ntwo\nthree");
expect(chunked.getText()).toBe(single.getText());
expect(chunked.getCursor()).toEqual(single.getCursor());
});
it("processes paste remainders iteratively, applying every trailing paste and keystroke", () => {
const editor = new Editor(defaultEditorTheme);
// One read carrying two complete pastes plus trailing typed text: the
// remainder after each paste loops back through input handling.
editor.handleInput("\x1b[200~ab\x1b[201~\x1b[200~cd\x1b[201~ef");
expect(editor.getText()).toBe("abcdef");
expect(editor.getCursor()).toEqual({ line: 0, col: 6 });
});
it("handles a long train of pastes in one read without recursing per remainder", () => {
const editor = new Editor(defaultEditorTheme);
editor.handleInput("\x1b[200~x\x1b[201~".repeat(2000));
expect(editor.getText()).toBe("x".repeat(2000));
});
it("inserts a plain printable run identically to per-scalar delivery", () => {
const run = "The quick brown fox 123 -_. naïve 😀 path";
const bulk = new Editor(defaultEditorTheme);
bulk.handleInput(run);
const perChar = new Editor(defaultEditorTheme);
for (const ch of run) perChar.handleInput(ch);
expect(bulk.getText()).toBe(perChar.getText());
expect(bulk.getCursor()).toEqual(perChar.getCursor());
});
it("keeps escape sequences interleaved with printable runs on the dispatch path", () => {
const editor = new Editor(defaultEditorTheme);
editor.handleInput("abc");
editor.handleInput("\x1b[D"); // Left
editor.handleInput("XY"); // bulk run lands before "c"
expect(editor.getText()).toBe("abXYc");
expect(editor.getCursor()).toEqual({ line: 0, col: 4 });
});
it("opens @ autocomplete when the trigger arrives inside a bulk printable run", async () => {
const editor = new Editor(defaultEditorTheme);
const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers<void>();
editor.setAutocompleteProvider({
async getSuggestions() {
return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" };
},
applyCompletion(lines, cursorLine, cursorCol) {
return { lines, cursorLine, cursorCol };
},
});
editor.onAutocompleteUpdate = resolveAutocompleteUpdated;
editor.handleInput("see @sr"); // one bulk run ending in an @-token
await autocompleteUpdated;
expect(editor.isShowingAutocomplete()).toBe(true);
});
it("opens @ autocomplete after a bracketed paste ending in a trigger token", async () => {
const editor = new Editor(defaultEditorTheme);
const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers<void>();
editor.setAutocompleteProvider({
async getSuggestions() {
return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" };
},
applyCompletion(lines, cursorLine, cursorCol) {
return { lines, cursorLine, cursorCol };
},
});
editor.onAutocompleteUpdate = resolveAutocompleteUpdated;
editor.handleInput("\x1b[200~see @sr\x1b[201~");
await autocompleteUpdated;
expect(editor.isShowingAutocomplete()).toBe(true);
});
});
describe("Korean NFC paste normalization", () => {
// macOS Finder drag-drops/Copy-As-Pathname emit Korean filenames as
// NFD (decomposed) — e.g. `화` becomes `ᄒ`(U+1112) + `ᅪ`(U+116A).
@@ -243,4 +243,87 @@ describe("Markdown incremental streaming lex (E2)", () => {
expect(streamLines).toEqual(renderCold(crlf.slice(0, len), 60));
}
});
// Closed-list lookahead: a "\n\n" boundary directly after a list token is
// freezable iff the tail cannot start a continuation item of that list
// (same bullet char, or 1-9 digits + same delimiter — marked's
// listItemRegex). These corpora cross list/non-list and
// list/incompatible-list boundaries; the divergence (and the freeze
// opportunity) is phase-sensitive, so each runs at step=1 and the
// production reveal granularity (step=3).
it("bullet list followed by a paragraph grows byte-identically", () => {
const doc =
"- alpha item with words\n- beta item with words\n- gamma item\n\n" +
"Closing paragraph that keeps streaming additional words to the end.";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
assertIdenticalGrowthTransient(doc, 60, 3);
});
it("bullet list followed by a different-marker list stays two lists", () => {
const doc = "- alpha\n- beta\n\n* starred one\n* starred two\n\n+ plus one\n+ plus two";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
});
it("ordered list followed by a paren-delimited list stays two lists", () => {
const doc = "1. dot one\n2. dot two\n\n1) paren one\n2) paren two";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
assertIdenticalGrowthTransient(doc, 60, 3);
});
it("list followed by blockquote grows byte-identically", () => {
const doc = "- alpha\n- beta\n\n> quoted line one with words\n> quoted line two here";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
});
it("list followed by heading grows byte-identically", () => {
const doc = "1. one\n2. two\n\n# Heading after the list\n\nTail prose keeps going on.";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
});
it("list followed by fenced code grows byte-identically", () => {
const doc = "- alpha\n- beta\n\n```ts\nconst x = compute(a, b);\nreturn x;\n```\n\ntail text";
assertIdenticalGrowth(doc, 60, 1);
assertIdenticalGrowth(doc, 60, 3);
});
it("a same-marker list across a blank line still merges while growing", () => {
const bullets = "- a\n- b\n\n- c\n- d";
assertIdenticalGrowth(bullets, 60, 1);
assertIdenticalGrowth(bullets, 60, 3);
});
it("a list closed by a paragraph freezes at the boundary (streaming perf gate)", () => {
// The lookahead must actually fire here: the tail after the blank line
// is a paragraph, which cannot continue a `-` list, so the rendered
// list rows become settled (frozen prefix) on the transient path.
const doc = "- alpha\n- beta\n- gamma\n\nClosing paragraph after the list keeps going.";
const streaming = new Markdown("", 0, 0, THEME);
streaming.transientRenderCache = true;
clearRenderCache();
streaming.setText(doc);
const streamLines = streaming.render(60);
expect(streamLines).toEqual(renderCold(doc, 60));
expect(streaming.getLastRenderSettledRows()).toBeGreaterThan(0);
});
it("a document that is one still-growing list never freezes mid-list", () => {
// No (b)-style intra-list freezing shipped: loose/tight and ordered
// renumbering are whole-list properties, so no prefix of an open list
// is byte-stable. Settled rows must stay 0 for a pure-list document.
const doc = "- one two three\n- four five six\n\n- seven eight nine";
const streaming = new Markdown("", 0, 0, THEME);
streaming.transientRenderCache = true;
for (let len = 1; len <= doc.length; len += 1) {
clearRenderCache();
streaming.setText(doc.slice(0, len));
const streamLines = streaming.render(60);
expect(streamLines).toEqual(renderCold(doc.slice(0, len), 60));
expect(streaming.getLastRenderSettledRows()).toBe(0);
}
});
});
+154 -2
View File
@@ -1,6 +1,13 @@
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
import { stripVTControlCharacters } from "node:util";
import { clearRenderCache, Markdown, renderInlineMarkdown } from "@oh-my-pi/pi-tui/components/markdown";
import {
autolinkSchemeScanIndex,
clearRenderCache,
Markdown,
mathStartIndex,
renderInlineMarkdown,
urlTokenPossible,
} from "@oh-my-pi/pi-tui/components/markdown";
import { setTerminalTextSizing, TERMINAL } from "@oh-my-pi/pi-tui/terminal-capabilities";
import { type Component, TUI } from "@oh-my-pi/pi-tui/tui";
import { visibleWidth } from "@oh-my-pi/pi-tui/utils";
@@ -1940,9 +1947,11 @@ describe("Markdown.render reference stability", () => {
});
it("does not share oversized renders through the L2 cache", () => {
// Fixture must exceed RENDER_CACHE_MAX_ENTRY_SIZE (256 KiB of rendered
// lines) so the entry is rejected and each render owns its array.
const width = 80;
const paragraph = `cache-budget sentinel ${"x".repeat(120)}`;
const largeText = Array.from({ length: 160 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n");
const largeText = Array.from({ length: 1400 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n");
const first = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width);
const second = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width);
@@ -2319,3 +2328,146 @@ describe("Math rendering", () => {
expect(lines[fxIdx + 1]).toContain("x > 0");
});
});
describe("inline start()/url-gate scanners (perf rewrites)", () => {
// The hand-rolled scanners replaced regex scans that marked runs on the
// remaining source at every inline position. They must return exactly what
// the old regexes returned for every input.
const OLD_MATH_START = /\$|\\\(|\\\[/;
const OLD_AUTOLINK_SCAN = /www\.|https?:\/\/|ftp:\/\//i;
// marked's bundled GFM inline url rule (verbatim, no flags).
const GFM_URL_REGEX =
/^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9-]+\.?)+[^\s<]*|^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/;
const fixtures = [
"",
"plain prose with no candidates at all",
"$x$ math first",
"prose then $inline$ math",
"prose then \\(paren\\) math",
"prose then \\[bracket\\] math",
"\\( before $ dollar",
"$ before \\( paren",
"backslash only \\ then ( apart",
"ends with backslash \\",
"ends with dollar $",
"www.example.com leading",
"see www.example.com mid-string",
"see WWW.EXAMPLE.COM upper",
"mixed WwW.case.com scan",
"http://example.com leading",
"prose http://example.com mid",
"prose HTTPS://EXAMPLE.COM upper",
"HtTpS://mixed.example",
"ftp://files.example mid ftp",
"prose FTP://FILES.EXAMPLE",
"ftps:// is not ftp:// until here ftp://x",
"wwww.overlap.example",
"hhttp://overlap.example",
"http:/ missing slash then https://real.example",
"www without dot www. with dot",
"w h f teaser chars but no scheme",
"user@example.com email",
"prose user.name+tag@example.co.uk",
"trailing at sign only@ ",
"@leading-at no local part",
"a".repeat(400), // long identifier run, no @
`${"a".repeat(400)}@example.com`, // long local part (past gate scan limit)
"short@x",
"dots...and+plus_under-score@host.tld",
];
it("mathStartIndex matches the old /\\$|\\\\\\(|\\\\\\[/ scan on every fixture", () => {
for (const src of fixtures) {
const m = OLD_MATH_START.exec(src);
expect(mathStartIndex(src)).toBe(m ? m.index : undefined);
}
});
it("autolinkSchemeScanIndex matches the old /www\\.|https?:\\/\\/|ftp:\\/\\//i scan on every fixture", () => {
for (const src of fixtures) {
const m = OLD_AUTOLINK_SCAN.exec(src);
expect(autolinkSchemeScanIndex(src)).toBe(m ? m.index : undefined);
}
});
it("urlTokenPossible is conservative: never false when the GFM url regex matches", () => {
for (const src of fixtures) {
if (GFM_URL_REGEX.test(src)) {
expect(urlTokenPossible(src)).toBeTrue();
}
}
// And it actually gates: plain prose with no scheme/email head is rejected.
expect(urlTokenPossible("plain prose, nothing linkable here")).toBeFalse();
expect(urlTokenPossible("@leading-at no local part")).toBeFalse();
});
it("gated tokenizer still autolinks urls and emails end-to-end", () => {
const rendered = renderInlineMarkdown("see https://example.com and mail user@example.com now", {
...defaultMarkdownTheme,
link: (text: string) => `<L>${text}</L>`,
});
const plain = stripVTControlCharacters(rendered);
expect(plain).toContain("<L>https://example.com</L>");
expect(plain).toContain("<L>user@example.com</L>");
});
});
describe("windowed lexing (documents past WINDOWED_LEX_MIN_BYTES)", () => {
// Large documents are lexed in bounded windows because Bun's regex engine
// rescans the whole remaining source for marked's `^`-anchored block rules.
// Every construct below straddles window cuts; a bad cut is visible in the
// rendered output.
afterEach(() => clearRenderCache());
const filler = (label: string, lines: number) =>
Array.from({ length: lines }, (_, i) => `${label} paragraph ${i} with enough prose to fill a window.`).join(
"\n\n",
);
const plain = (text: string, width = 100) =>
new Markdown(text, 0, 0, defaultMarkdownTheme)
.render(width)
.map(line => stripVTControlCharacters(line).trimEnd());
it("resolves a reference definition that lands in a later window", () => {
const doc = `Follow [the label][ref] first.\n\n${filler("body", 400)}\n\n[ref]: https://example.com/late\n`;
expect(doc.length).toBeGreaterThan(16 * 1024);
const rendered = plain(doc, 120);
// The reflink resolved: marked emitted a link token (rendered as
// `label (href)`), so the raw `[label][ref]` syntax is gone and the
// definition line itself produced no output block of its own.
expect(rendered[0]).toBe("Follow the label (https://example.com/late) first.");
expect(rendered.filter(line => line.includes("https://example.com/late"))).toHaveLength(1);
});
it("keeps a fenced block longer than one window intact", () => {
const code = Array.from({ length: 200 }, (_, i) => `const value${i} = ${i};`).join("\n");
const doc = `${filler("intro", 300)}\n\n\`\`\`ts\n${code}\n\`\`\`\n\n${filler("outro", 20)}`;
expect(code.length).toBeGreaterThan(2 * 1024);
const rendered = plain(doc);
// Exactly one fence pair: a window cut inside the block would close and
// reopen it (or spill code lines into prose).
expect(rendered.filter(line => line.trimStart().startsWith("```"))).toHaveLength(2);
const first = rendered.findIndex(line => line.includes("const value0 = 0;"));
expect(first).toBeGreaterThan(-1);
for (let i = 0; i < 200; i++) {
expect(rendered[first + i]).toContain(`const value${i} = ${i};`);
}
});
it("numbers an ordered list continuously across window cuts", () => {
const items = Array.from({ length: 400 }, (_, i) => `${i + 1}. item ${i} padded with extra words to add bytes`);
const doc = `${filler("intro", 60)}\n\n${items.join("\n")}\n`;
expect(doc.length).toBeGreaterThan(16 * 1024);
const rendered = plain(doc, 120);
for (const n of [1, 137, 400]) {
expect(rendered.some(line => line.includes(`${n}. item ${n - 1} `))).toBe(true);
}
// A window cut that restarted the list would renumber later items.
expect(rendered.filter(line => line.includes(" 1. item 0 ")).length).toBeLessThanOrEqual(1);
});
});