perf(tui): optimized markdown lexing, editor input, and cursor updates
- Optimized markdown streaming lexing, tokenizers, and render cache limits. - Improved editor input handling with bulk printable fast paths and iterative chunk processing. - Added cursor visibility tracking and deduplication to prevent redundant terminal escape sequences. - Added benchmarks and unit tests covering markdown streaming, incremental lexing, and editor fast paths.
This commit is contained in:
@@ -2,6 +2,16 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added bulk-input fast path and iterative processing for bracketed paste in the editor
|
||||
- Added windowed incremental lexing for large markdown documents
|
||||
|
||||
### Changed
|
||||
|
||||
- Optimized markdown URL tokenizer gate, inline math start scan, and autolink scheme scan for performance
|
||||
- Deduplicated terminal cursor-visibility writes to skip redundant escape sequences
|
||||
|
||||
## [17.1.6] - 2026-07-27
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -528,8 +528,8 @@ interface Terminal {
|
||||
get columns(): number;
|
||||
get rows(): number;
|
||||
moveBy(lines: number): void;
|
||||
hideCursor(): void;
|
||||
showCursor(): void;
|
||||
hideCursor(force?: boolean): void;
|
||||
showCursor(force?: boolean): void;
|
||||
clearLine(): void;
|
||||
clearFromCursor(): void;
|
||||
clearScreen(): void;
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
/**
|
||||
* Streaming markdown render benchmark.
|
||||
*
|
||||
* Simulates a model streaming a long markdown message into one reused
|
||||
* `Markdown` component (the interactive-mode hot path): the text grows in
|
||||
* fixed-size deltas and the component re-renders after each delta.
|
||||
*
|
||||
* Exercises the profile hotspots from the 2026-07 capture: marked's GFM `url`
|
||||
* tokenizer (73.3% self), inline extension `start()` scans, emStrong, and the
|
||||
* streaming stable-prefix freeze (`#freezeStablePrefix`).
|
||||
*
|
||||
* Run: bun packages/tui/bench/markdown-stream.ts
|
||||
*/
|
||||
import { clearRenderCache, Markdown } from "../src/components/markdown";
|
||||
import { defaultMarkdownTheme } from "../test/test-themes";
|
||||
|
||||
const WIDTH = 100;
|
||||
const DELTA = 64; // chars revealed per streaming step
|
||||
|
||||
// --- Fixtures -------------------------------------------------------------
|
||||
|
||||
/** Long bullet list — the shape that defeats prefix freezing (list guard). */
|
||||
function bulletList(items: number): string {
|
||||
const lines: string[] = [];
|
||||
for (let i = 0; i < items; i++) {
|
||||
lines.push(
|
||||
`- \`packages/tui/src/components/markdown_component_${i}.ts\` handles the ` +
|
||||
`stable_prefix_freeze_path_${i} and re-lexes only the unfrozen tail, see ` +
|
||||
`https://github.com/can1357/oh-my-pi/issues/${1000 + i} for details on token_${i}.`,
|
||||
);
|
||||
}
|
||||
return `${lines.join("\n")}\n\n`;
|
||||
}
|
||||
|
||||
/** Prose dense in email-branch pathology: long `[A-Za-z0-9._+-]+` runs with no `@`. */
|
||||
function identifierProse(paragraphs: number): string {
|
||||
const parts: string[] = [];
|
||||
for (let i = 0; i < paragraphs; i++) {
|
||||
parts.push(
|
||||
`The resolver maps session_listing.scan_session_file.header_cache_v${i} onto ` +
|
||||
`auth_broker.remote_store.filter_usage_reports_${i} while user${i}@example.com and ` +
|
||||
`www.example${i}.org stay autolinked; identifiers like RENDER_CACHE_MAX_ENTRY_SIZE_${i} ` +
|
||||
`and freeze.stable.prefix.tokens.v${i} must parse as plain *text* with **no** backtracking.`,
|
||||
);
|
||||
}
|
||||
return `${parts.join("\n\n")}\n\n`;
|
||||
}
|
||||
|
||||
function fences(count: number): string {
|
||||
const parts: string[] = [];
|
||||
for (let i = 0; i < count; i++) {
|
||||
parts.push("```ts\nconst x_" + i + " = await fetch(\"https://api.example.com/v1/usage\");\n```\n");
|
||||
}
|
||||
return `${parts.join("\n")}\n`;
|
||||
}
|
||||
|
||||
const DOC = identifierProse(20) + bulletList(120) + fences(8) + identifierProse(20) + bulletList(80);
|
||||
|
||||
// --- Bench ----------------------------------------------------------------
|
||||
|
||||
function streamOnce(text: string): number {
|
||||
clearRenderCache();
|
||||
const component = new Markdown("", 0, 0, defaultMarkdownTheme);
|
||||
component.transientRenderCache = true;
|
||||
const start = Bun.nanoseconds();
|
||||
for (let len = DELTA; len < text.length; len += DELTA) {
|
||||
component.setText(text.slice(0, len));
|
||||
component.render(WIDTH);
|
||||
}
|
||||
component.setText(text);
|
||||
component.render(WIDTH);
|
||||
return (Bun.nanoseconds() - start) / 1e6;
|
||||
}
|
||||
|
||||
function coldOnce(text: string): number {
|
||||
clearRenderCache();
|
||||
const start = Bun.nanoseconds();
|
||||
new Markdown(text, 0, 0, defaultMarkdownTheme).render(WIDTH);
|
||||
return (Bun.nanoseconds() - start) / 1e6;
|
||||
}
|
||||
|
||||
console.log(`doc: ${DOC.length} chars, ${Math.ceil(DOC.length / DELTA)} streaming steps, width ${WIDTH}`);
|
||||
// Warmup (JIT + regex compilation)
|
||||
streamOnce(DOC.slice(0, 4096));
|
||||
|
||||
const cold = coldOnce(DOC);
|
||||
console.log(`cold full render: ${cold.toFixed(1)}ms`);
|
||||
|
||||
const runs: number[] = [];
|
||||
for (let i = 0; i < 3; i++) runs.push(streamOnce(DOC));
|
||||
runs.sort((a, b) => a - b);
|
||||
console.log(`streamed render (${runs.length} runs): min ${runs[0]!.toFixed(1)}ms, median ${runs[1]!.toFixed(1)}ms`);
|
||||
@@ -6,8 +6,8 @@ import {
|
||||
midPromptSkillTokenMatches,
|
||||
} from "../autocomplete";
|
||||
import { BracketedPasteHandler, decodeReencodedPasteControls } from "../bracketed-paste";
|
||||
import { getKeybindings, type KeybindingsManager } from "../keybindings";
|
||||
import { extractPrintableText, matchesKey } from "../keys";
|
||||
import { canonicalKeyId, getKeybindings, type KeybindingsManager } from "../keybindings";
|
||||
import { extractPrintableText, matchesKey, parseKey } from "../keys";
|
||||
import { KillRing } from "../kill-ring";
|
||||
import type { SymbolTheme } from "../symbols";
|
||||
import { type Component, CURSOR_MARKER, type Focusable } from "../tui";
|
||||
@@ -341,6 +341,17 @@ function maxSegmentVisualCol(text: string, isLastSegment: boolean): number {
|
||||
return isLastSegment ? total : Math.max(0, total - lastWidth);
|
||||
}
|
||||
|
||||
/** True when every code unit is plain printable text: no C0 controls (so no
|
||||
* ESC/CR/LF/TAB), no DEL, no C1 range — the same set `extractPrintableText`
|
||||
* rejects. Such a run can never encode a key sequence. */
|
||||
function isPlainTextRun(data: string): boolean {
|
||||
for (let i = 0; i < data.length; i++) {
|
||||
const code = data.charCodeAt(i);
|
||||
if (code < 0x20 || code === 0x7f || (code >= 0x80 && code <= 0x9f)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const DEFAULT_PAGE_SCROLL_LINES = 10;
|
||||
|
||||
const MAX_UNDO_STACK = 100;
|
||||
@@ -1126,12 +1137,30 @@ export class Editor implements Component, Focusable {
|
||||
}
|
||||
|
||||
handleInput(data: string): void {
|
||||
// Iterative, not recursive: the bytes trailing a completed bracketed
|
||||
// paste (which may themselves contain further pastes) loop back here,
|
||||
// so a fragmented paste stream can never grow the call stack.
|
||||
let next: string | undefined = data;
|
||||
while (next !== undefined && next.length > 0) {
|
||||
next = this.#handleInputChunk(next);
|
||||
}
|
||||
}
|
||||
|
||||
/** Process one input chunk. Returns the unconsumed tail of a completed paste, if any. */
|
||||
#handleInputChunk(data: string): string | undefined {
|
||||
const kb = getKeybindings();
|
||||
// Parse the sequence once; every binding probe below is then a set
|
||||
// lookup instead of re-parsing `data` per probe (~35 probes per key).
|
||||
const parsedKey = parseKey(data);
|
||||
const canonical = parsedKey === undefined ? undefined : canonicalKeyId(parsedKey);
|
||||
|
||||
// Handle character jump mode (awaiting next character to jump to)
|
||||
if (this.#jumpMode !== null) {
|
||||
// Cancel if the hotkey is pressed again
|
||||
if (kb.matches(data, "tui.editor.jumpForward") || kb.matches(data, "tui.editor.jumpBackward")) {
|
||||
if (
|
||||
kb.matchesCanonical(canonical, "tui.editor.jumpForward") ||
|
||||
kb.matchesCanonical(canonical, "tui.editor.jumpBackward")
|
||||
) {
|
||||
this.#jumpMode = null;
|
||||
return;
|
||||
}
|
||||
@@ -1154,12 +1183,23 @@ export class Editor implements Component, Focusable {
|
||||
if (paste.pasteContent !== undefined) {
|
||||
this.#handlePaste(paste.pasteContent);
|
||||
if (paste.remaining.length > 0) {
|
||||
this.handleInput(paste.remaining);
|
||||
return paste.remaining;
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Bulk printable fast path: a multi-scalar run of plain text (paste
|
||||
// remainder, batched stdin) parses to no key, so no binding probe or
|
||||
// special-key branch below can consume it — it always falls through to
|
||||
// one #insertCharacter call. Take that path directly and skip the
|
||||
// dispatch cascade. Runs containing ESC or control bytes (including
|
||||
// \r/\n) keep the full path: those bytes carry key semantics.
|
||||
if (canonical === undefined && data.length > 1 && isPlainTextRun(data)) {
|
||||
this.#insertCharacter(data);
|
||||
return;
|
||||
}
|
||||
|
||||
// Handle special key combinations first
|
||||
|
||||
// Ctrl+C is reserved by parent components for app-level handling.
|
||||
@@ -1170,7 +1210,7 @@ export class Editor implements Component, Focusable {
|
||||
}
|
||||
|
||||
// Undo
|
||||
if (kb.matches(data, "tui.editor.undo")) {
|
||||
if (kb.matchesCanonical(canonical, "tui.editor.undo")) {
|
||||
this.#applyUndo();
|
||||
return;
|
||||
}
|
||||
@@ -1178,26 +1218,26 @@ export class Editor implements Component, Focusable {
|
||||
// Handle autocomplete special keys first (but don't block other input)
|
||||
if (this.#autocompleteState && this.#autocompleteList) {
|
||||
// Escape - cancel autocomplete
|
||||
if (kb.matches(data, "tui.select.cancel")) {
|
||||
if (kb.matchesCanonical(canonical, "tui.select.cancel")) {
|
||||
this.#cancelAutocomplete(true);
|
||||
return;
|
||||
}
|
||||
// Let the autocomplete list handle navigation and selection
|
||||
else if (
|
||||
kb.matches(data, "tui.select.up") ||
|
||||
kb.matches(data, "tui.select.down") ||
|
||||
kb.matches(data, "tui.select.pageUp") ||
|
||||
kb.matches(data, "tui.select.pageDown") ||
|
||||
kb.matches(data, "tui.input.submit") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.up") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.down") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.pageUp") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.pageDown") ||
|
||||
kb.matchesCanonical(canonical, "tui.input.submit") ||
|
||||
data === "\n" ||
|
||||
kb.matches(data, "tui.input.tab")
|
||||
kb.matchesCanonical(canonical, "tui.input.tab")
|
||||
) {
|
||||
// Only pass navigation keys to the list, not Enter/Tab (we handle those directly)
|
||||
if (
|
||||
kb.matches(data, "tui.select.up") ||
|
||||
kb.matches(data, "tui.select.down") ||
|
||||
kb.matches(data, "tui.select.pageUp") ||
|
||||
kb.matches(data, "tui.select.pageDown")
|
||||
kb.matchesCanonical(canonical, "tui.select.up") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.down") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.pageUp") ||
|
||||
kb.matchesCanonical(canonical, "tui.select.pageDown")
|
||||
) {
|
||||
this.#autocompleteList.handleInput(data);
|
||||
this.onAutocompleteUpdate?.();
|
||||
@@ -1205,7 +1245,7 @@ export class Editor implements Component, Focusable {
|
||||
}
|
||||
|
||||
// If Tab was pressed, always apply the selection
|
||||
if (kb.matches(data, "tui.input.tab")) {
|
||||
if (kb.matchesCanonical(canonical, "tui.input.tab")) {
|
||||
const selected = this.#autocompleteList.getSelectedItem();
|
||||
// Check for stale autocomplete state due to buffer edits since last refresh
|
||||
// (destructive keys or paste can outrun the debounced update).
|
||||
@@ -1249,7 +1289,7 @@ export class Editor implements Component, Focusable {
|
||||
// If Enter was pressed on a submitted slash command (not an absolute-path
|
||||
// completion sharing the leading-slash prefix), apply and submit.
|
||||
if (
|
||||
(kb.matches(data, "tui.input.submit") || data === "\n") &&
|
||||
(kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") &&
|
||||
findLeadingSlashCommandStart(this.#autocompletePrefix) !== null &&
|
||||
this.#isInSubmittedSlashCommandContext() &&
|
||||
!this.#selectedCompletionIsPath()
|
||||
@@ -1281,7 +1321,7 @@ export class Editor implements Component, Focusable {
|
||||
// Don't return - fall through to submission logic
|
||||
}
|
||||
// Otherwise, apply the completion without submitting the surrounding draft.
|
||||
else if (kb.matches(data, "tui.input.submit") || data === "\n") {
|
||||
else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") {
|
||||
const selected = this.#autocompleteList.getSelectedItem();
|
||||
// Check for stale autocomplete state due to buffer edits since last refresh.
|
||||
const currentLine = this.#state.lines[this.#state.cursorLine] ?? "";
|
||||
@@ -1321,37 +1361,37 @@ export class Editor implements Component, Focusable {
|
||||
}
|
||||
|
||||
// Tab key - context-aware completion (but not when already autocompleting)
|
||||
if (kb.matches(data, "tui.input.tab") && !this.#autocompleteState) {
|
||||
if (kb.matchesCanonical(canonical, "tui.input.tab") && !this.#autocompleteState) {
|
||||
this.#handleTabCompletion();
|
||||
return;
|
||||
}
|
||||
|
||||
// Continue with rest of input handling
|
||||
// Delete to end of line
|
||||
if (kb.matches(data, "tui.editor.deleteToLineEnd")) {
|
||||
if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineEnd")) {
|
||||
this.#deleteToEndOfLine();
|
||||
}
|
||||
// Delete to start of line
|
||||
else if (kb.matches(data, "tui.editor.deleteToLineStart")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.deleteToLineStart")) {
|
||||
this.#deleteToStartOfLine();
|
||||
}
|
||||
// Delete word backward. Registry defaults cover ctrl+w, alt+backspace,
|
||||
// ctrl+backspace, and super+alt+backspace (Ghostty on macOS reports
|
||||
// Option+Backspace as super+alt — kitty mod 11, see #2064).
|
||||
else if (kb.matches(data, "tui.editor.deleteWordBackward")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordBackward")) {
|
||||
this.#deleteWordBackwards();
|
||||
}
|
||||
// Delete word forward. Registry defaults cover alt+d/alt+delete and their
|
||||
// super+alt variants for the same Ghostty quirk.
|
||||
else if (kb.matches(data, "tui.editor.deleteWordForward")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.deleteWordForward")) {
|
||||
this.#deleteWordForwards();
|
||||
}
|
||||
// Yank from kill ring
|
||||
else if (kb.matches(data, "tui.editor.yank")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.yank")) {
|
||||
this.#yankFromKillRing();
|
||||
}
|
||||
// Yank-pop (cycle kill ring)
|
||||
else if (kb.matches(data, "tui.editor.yankPop")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.yankPop")) {
|
||||
this.#yankPop();
|
||||
}
|
||||
// Ctrl+A - Move to start of line
|
||||
@@ -1376,7 +1416,7 @@ export class Editor implements Component, Focusable {
|
||||
matchesKey(data, "ctrl+enter") || // Ctrl+Enter (Kitty/modifyOtherKeys, including lock bits/keypad Enter)
|
||||
data === "\x1b\r" || // Option+Enter in some terminals (legacy)
|
||||
data === "\x1b[13;2~" || // Shift+Enter in some terminals (legacy format)
|
||||
kb.matches(data, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits)
|
||||
kb.matchesCanonical(canonical, "tui.input.newLine") || // Shift+Enter (Kitty protocol, handles lock bits)
|
||||
(data.length > 1 && data.includes("\x1b") && data.includes("\r")) ||
|
||||
(data === "\n" && data.length === 1) // Shift+Enter from iTerm2 mapping
|
||||
) {
|
||||
@@ -1388,7 +1428,7 @@ export class Editor implements Component, Focusable {
|
||||
this.#addNewLine();
|
||||
}
|
||||
// Plain Enter - submit (handles both legacy \r and Kitty protocol with lock bits)
|
||||
else if (kb.matches(data, "tui.input.submit") || data === "\n") {
|
||||
else if (kb.matchesCanonical(canonical, "tui.input.submit") || data === "\n") {
|
||||
// If submit is disabled, do nothing
|
||||
if (this.disableSubmit) {
|
||||
return;
|
||||
@@ -1429,40 +1469,40 @@ export class Editor implements Component, Focusable {
|
||||
this.#submitValue();
|
||||
}
|
||||
// Backspace (including Shift+Backspace)
|
||||
else if (kb.matches(data, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharBackward") || matchesKey(data, "shift+backspace")) {
|
||||
this.#handleBackspace();
|
||||
}
|
||||
// Line navigation shortcuts (Home/End keys)
|
||||
else if (kb.matches(data, "tui.editor.cursorLineStart")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineStart")) {
|
||||
this.#moveToLineStart();
|
||||
} else if (kb.matches(data, "tui.editor.cursorLineEnd")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorLineEnd")) {
|
||||
this.#moveToLineEnd();
|
||||
}
|
||||
// Page navigation (PageUp/PageDown): page the editor viewport only. On a
|
||||
// short draft this is a no-op — it never steps prompt history (that stays
|
||||
// on Up/Down), so an idle empty editor swallows the keys instead of
|
||||
// surprising the user by loading the previous prompt (#4754).
|
||||
else if (kb.matches(data, "tui.editor.pageUp")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.pageUp")) {
|
||||
this.#pageScroll(-1);
|
||||
} else if (kb.matches(data, "tui.editor.pageDown")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.pageDown")) {
|
||||
this.#pageScroll(1);
|
||||
}
|
||||
// Forward delete (Fn+Backspace or Delete key, including Shift+Delete)
|
||||
else if (kb.matches(data, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.deleteCharForward") || matchesKey(data, "shift+delete")) {
|
||||
this.#handleForwardDelete();
|
||||
}
|
||||
// Word navigation (Option/Alt + Arrow or Ctrl + Arrow)
|
||||
else if (kb.matches(data, "tui.editor.cursorWordLeft")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordLeft")) {
|
||||
// Word left
|
||||
this.#resetKillSequence();
|
||||
this.#moveWordBackwards();
|
||||
} else if (kb.matches(data, "tui.editor.cursorWordRight")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorWordRight")) {
|
||||
// Word right
|
||||
this.#resetKillSequence();
|
||||
this.#moveWordForwards();
|
||||
}
|
||||
// Arrow keys
|
||||
else if (kb.matches(data, "tui.editor.cursorUp")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.cursorUp")) {
|
||||
// Up - history navigation or cursor movement
|
||||
if (this.#isEditorEmpty()) {
|
||||
this.#navigateHistory(-1); // Start browsing history
|
||||
@@ -1474,7 +1514,7 @@ export class Editor implements Component, Focusable {
|
||||
} else {
|
||||
this.#moveCursor(-1, 0); // Cursor movement (within text or history entry)
|
||||
}
|
||||
} else if (kb.matches(data, "tui.editor.cursorDown")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorDown")) {
|
||||
// Down - history navigation or cursor movement
|
||||
if (this.#historyIndex > -1 && this.#isOnLastVisualLine()) {
|
||||
this.#navigateHistory(1); // Navigate to newer history entry or clear
|
||||
@@ -1484,10 +1524,10 @@ export class Editor implements Component, Focusable {
|
||||
} else {
|
||||
this.#moveCursor(1, 0); // Cursor movement (within text or history entry)
|
||||
}
|
||||
} else if (kb.matches(data, "tui.editor.cursorRight")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorRight")) {
|
||||
// Right
|
||||
this.#moveCursor(0, 1);
|
||||
} else if (kb.matches(data, "tui.editor.cursorLeft")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.cursorLeft")) {
|
||||
// Left
|
||||
this.#moveCursor(0, -1);
|
||||
}
|
||||
@@ -1496,9 +1536,9 @@ export class Editor implements Component, Focusable {
|
||||
this.#insertCharacter(" ");
|
||||
}
|
||||
// Character jump mode triggers
|
||||
else if (kb.matches(data, "tui.editor.jumpForward")) {
|
||||
else if (kb.matchesCanonical(canonical, "tui.editor.jumpForward")) {
|
||||
this.#jumpMode = "forward";
|
||||
} else if (kb.matches(data, "tui.editor.jumpBackward")) {
|
||||
} else if (kb.matchesCanonical(canonical, "tui.editor.jumpBackward")) {
|
||||
this.#jumpMode = "backward";
|
||||
}
|
||||
// Printable keystrokes, including Kitty CSI-u text-producing sequences.
|
||||
@@ -1630,6 +1670,15 @@ export class Editor implements Component, Focusable {
|
||||
return this.#state.lines.join("\n");
|
||||
}
|
||||
|
||||
/** Whether the buffer text equals `value`, without `getText()`'s full join —
|
||||
* O(1) for the hot per-keystroke probes against short single-line values. */
|
||||
textEquals(value: string): boolean {
|
||||
const lines = this.#state.lines;
|
||||
if (lines.length === 1) return lines[0] === value;
|
||||
if (value.indexOf("\n") === -1) return false;
|
||||
return this.getText() === value;
|
||||
}
|
||||
|
||||
#expandPasteMarkers(text: string): string {
|
||||
let result = text;
|
||||
for (const [pasteId, pasteContent] of this.#pastes) {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { LRUCache } from "lru-cache/raw";
|
||||
import { Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked";
|
||||
import { Lexer, Marked, type Token, Tokenizer, type TokenizerAndRendererExtension, type Tokens } from "marked";
|
||||
import { latexToBlock } from "../latex-block";
|
||||
import { inlineMathSpanEnd, isBareMathEnvironment, latexToUnicode } from "../latex-to-unicode";
|
||||
import type { SymbolTheme } from "../symbols";
|
||||
@@ -492,12 +492,25 @@ const customHrExtension: TokenizerAndRendererExtension = {
|
||||
},
|
||||
};
|
||||
|
||||
// Leftmost-match scan replacing /\$|\\\(|\\\[/ in mathExtension.start —
|
||||
// marked calls start() on the remaining source at every inline position, so
|
||||
// the regex alternation showed up in CPU profiles (part of a ~4.3% start()
|
||||
// tail). Three indexOf scans yield the identical leftmost index.
|
||||
/** @internal exported for tests — must stay index-identical to the old regex scan. */
|
||||
export function mathStartIndex(src: string): number | undefined {
|
||||
let best = src.indexOf("$");
|
||||
const paren = src.indexOf("\\(");
|
||||
if (paren !== -1 && (best === -1 || paren < best)) best = paren;
|
||||
const bracket = src.indexOf("\\[");
|
||||
if (bracket !== -1 && (best === -1 || bracket < best)) best = bracket;
|
||||
return best === -1 ? undefined : best;
|
||||
}
|
||||
|
||||
const mathExtension: TokenizerAndRendererExtension = {
|
||||
name: "math",
|
||||
level: "inline",
|
||||
start(src) {
|
||||
const m = /\$|\\\(|\\\[/.exec(src);
|
||||
return m ? m.index : undefined;
|
||||
return mathStartIndex(src);
|
||||
},
|
||||
tokenizer(src) {
|
||||
if (src.startsWith("$$")) {
|
||||
@@ -614,14 +627,63 @@ const mathEnvBlockExtension: TokenizerAndRendererExtension = {
|
||||
// tokenizer at a valid start. Candidates at a legal boundary fall through
|
||||
// (return undefined) to marked's own autolink handling unchanged.
|
||||
const AUTOLINK_SCHEME_REGEX = /^(?:www\.|https?:\/\/|ftp:\/\/)/i;
|
||||
const AUTOLINK_SCHEME_SCAN = /www\.|https?:\/\/|ftp:\/\//i;
|
||||
// Case-insensitive scheme scan replacing /www\.|https?:\/\/|ftp:\/\//i in
|
||||
// boundedAutolinkExtension.start — like mathStartIndex above, this runs on the
|
||||
// remaining source at every inline position (part of a ~4.3% CPU start() scan
|
||||
// tail in profiles). charCode-only: no allocation, no toLowerCase copies.
|
||||
// `| 32` lower-cases ASCII letters; `.`/`:`/`/` are compared exactly, matching
|
||||
// the regex's ASCII-only `i` semantics. charCodeAt past the end returns NaN,
|
||||
// which fails every comparison, so no explicit bounds checks are needed.
|
||||
function isAutolinkSchemeAt(src: string, i: number): boolean {
|
||||
const c = src.charCodeAt(i) | 32;
|
||||
if (c === 119 /* w */) {
|
||||
// www.
|
||||
return (
|
||||
(src.charCodeAt(i + 1) | 32) === 119 &&
|
||||
(src.charCodeAt(i + 2) | 32) === 119 &&
|
||||
src.charCodeAt(i + 3) === 46 /* . */
|
||||
);
|
||||
}
|
||||
if (c === 104 /* h */) {
|
||||
// http:// | https://
|
||||
if (
|
||||
(src.charCodeAt(i + 1) | 32) !== 116 /* t */ ||
|
||||
(src.charCodeAt(i + 2) | 32) !== 116 /* t */ ||
|
||||
(src.charCodeAt(i + 3) | 32) !== 112 /* p */
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
let j = i + 4;
|
||||
if ((src.charCodeAt(j) | 32) === 115 /* s */) j++;
|
||||
return src.charCodeAt(j) === 58 /* : */ && src.charCodeAt(j + 1) === 47 /* / */ && src.charCodeAt(j + 2) === 47;
|
||||
}
|
||||
if (c === 102 /* f */) {
|
||||
// ftp://
|
||||
return (
|
||||
(src.charCodeAt(i + 1) | 32) === 116 /* t */ &&
|
||||
(src.charCodeAt(i + 2) | 32) === 112 /* p */ &&
|
||||
src.charCodeAt(i + 3) === 58 /* : */ &&
|
||||
src.charCodeAt(i + 4) === 47 /* / */ &&
|
||||
src.charCodeAt(i + 5) === 47 /* / */
|
||||
);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** @internal exported for tests — must stay index-identical to the old regex scan. */
|
||||
export function autolinkSchemeScanIndex(src: string): number | undefined {
|
||||
for (let i = 0; i < src.length; i++) {
|
||||
const c = src.charCodeAt(i) | 32;
|
||||
if ((c === 119 || c === 104 || c === 102) && isAutolinkSchemeAt(src, i)) return i;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
const VALID_AUTOLINK_LEFT_BOUNDARY = /[\s*_~(]/;
|
||||
const boundedAutolinkExtension: TokenizerAndRendererExtension = {
|
||||
name: "boundedAutolink",
|
||||
level: "inline",
|
||||
start(src) {
|
||||
const m = AUTOLINK_SCHEME_SCAN.exec(src);
|
||||
return m ? m.index : undefined;
|
||||
return autolinkSchemeScanIndex(src);
|
||||
},
|
||||
tokenizer(src, tokens) {
|
||||
const match = AUTOLINK_SCHEME_REGEX.exec(src);
|
||||
@@ -639,6 +701,66 @@ markdownParser.use({
|
||||
extensions: [customHrExtension, mathBlockExtension, mathEnvBlockExtension, mathExtension, boundedAutolinkExtension],
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// GFM `url` tokenizer gate
|
||||
// ---------------------------------------------------------------------------
|
||||
// marked tries the bundled GFM `url` tokenizer at every inline tokenization
|
||||
// step, and its regex is expensive to FAIL: the email alternative
|
||||
// `^[A-Za-z0-9._+-]+(@)…` linearly consumes an identifier run, then backtracks
|
||||
// it one character at a time when no `@` follows. A 71414-sample / 1ms CPU
|
||||
// profile of the TUI put 73.3% of total CPU (74.9s of a 102s capture) inside
|
||||
// this single regex. The override below runs an O(bounded) charCode gate first
|
||||
// and only falls through to the built-in tokenizer — by returning `false`,
|
||||
// marked's tokenizer-override fallback contract — when a match is possible.
|
||||
//
|
||||
// Conservativeness argument. The built-in rule (no flags) is
|
||||
// /^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.?)+[^\s<]*
|
||||
// |^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/
|
||||
// Both alternatives are anchored, so any match constrains the head of src:
|
||||
// • Branch 1 requires src to start with `http://`, `https://`, `ftp://`
|
||||
// (scheme letters in any case) or lowercase `www.`. The gate accepts all of
|
||||
// these via isAutolinkSchemeAt(src, 0); it also over-accepts `WWW.`, a
|
||||
// harmless false positive (the built-in regex simply fails to match).
|
||||
// • Branch 2 requires src to start with one-or-more chars from
|
||||
// `[A-Za-z0-9._+-]` immediately followed by `@`. The gate scans that exact
|
||||
// class: if the run ends within URL_GATE_EMAIL_SCAN_LIMIT chars it accepts
|
||||
// iff the terminator is `@`; a run reaching the limit is accepted
|
||||
// unconditionally. Every src branch 2 can match is therefore accepted —
|
||||
// the gate never rejects a src the built-in regex would match.
|
||||
const URL_GATE_EMAIL_SCAN_LIMIT = 320;
|
||||
|
||||
/** @internal exported for tests — must never return false for a src the built-in url regex matches. */
|
||||
export function urlTokenPossible(src: string): boolean {
|
||||
if (isAutolinkSchemeAt(src, 0)) return true;
|
||||
let i = 0;
|
||||
while (i < URL_GATE_EMAIL_SCAN_LIMIT) {
|
||||
const c = src.charCodeAt(i);
|
||||
const isLocalChar =
|
||||
(c >= 97 && c <= 122) /* a-z */ ||
|
||||
(c >= 65 && c <= 90) /* A-Z */ ||
|
||||
(c >= 48 && c <= 57) /* 0-9 */ ||
|
||||
c === 46 /* . */ ||
|
||||
c === 95 /* _ */ ||
|
||||
c === 43 /* + */ ||
|
||||
c === 45; /* - */
|
||||
if (!isLocalChar) break;
|
||||
i++;
|
||||
}
|
||||
if (i === 0) return false;
|
||||
if (i >= URL_GATE_EMAIL_SCAN_LIMIT) return true; // over-long run: give up conservatively
|
||||
return src.charCodeAt(i) === 64 /* @ */;
|
||||
}
|
||||
|
||||
markdownParser.use({
|
||||
tokenizer: {
|
||||
url(src: string): Tokens.Link | undefined | false {
|
||||
// `false` → marked falls back to the built-in `url` tokenizer;
|
||||
// `undefined` → no url token here, built-in never runs.
|
||||
return urlTokenPossible(src) ? false : undefined;
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Module-level LRU render cache
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -649,8 +771,8 @@ markdownParser.use({
|
||||
// (Rust FFI) work for content/layout combinations already seen this session.
|
||||
|
||||
const RENDER_CACHE_MAX = 256; // sane cap: ~256 distinct message × width combos
|
||||
const RENDER_CACHE_MAX_SIZE = 512 * 1024;
|
||||
const RENDER_CACHE_MAX_ENTRY_SIZE = 32 * 1024;
|
||||
const RENDER_CACHE_MAX_SIZE = 4 * 1024 * 1024;
|
||||
const RENDER_CACHE_MAX_ENTRY_SIZE = 256 * 1024;
|
||||
const EMPTY_RENDER_LINES: readonly string[] = [];
|
||||
|
||||
interface RenderCacheEntry {
|
||||
@@ -684,6 +806,179 @@ function renderCacheEntrySize(entry: RenderCacheEntry): number {
|
||||
// over-matching is safe (it only costs the fast path), under-matching is not.
|
||||
const HAS_REF_DEF = /^ {0,3}\[(?:\\.|[^\]\\])+\]:/m;
|
||||
|
||||
// marked's list tokenizer (Tokenizer.list, marked v18) continues a list across
|
||||
// blank lines only when the remaining source matches
|
||||
// `listItemRegex(marker)` = `^( {0,3}${marker})((?:[\t ][^\n]*)?(?:\n|$))`,
|
||||
// where `marker` is the exact bullet char for unordered lists (`\${char}`) or
|
||||
// 1-9 digits plus the exact delimiter for ordered lists (`\d{1,9}\${delim}`).
|
||||
// The marker is derived from the list's FIRST item (`n = t[1].trim()`), which
|
||||
// sits at the start of a top-level list token's raw:
|
||||
const LIST_MARKER_RE = /^ {0,3}(?:([*+-])|\d{1,9}([.)]))/;
|
||||
|
||||
// Streaming-freeze equivalence invariant: lex(prefix) ++ lex(tail) must equal
|
||||
// lex(full text) — for the CURRENT text and for every append-only extension of
|
||||
// it, because a frozen prefix is sticky (it keeps being reused while the text
|
||||
// grows). At a blank-line (`\n\n`) cut directly after a top-level `list`
|
||||
// token, the only construct that can straddle the cut is a continuation item
|
||||
// of that list: marked consumed the blank line into the last item's raw and
|
||||
// re-ran `listItemRegex` at exactly `tailStart`, merging a same-marker item
|
||||
// into one renumbered loose list. The cut is safe only when that regex can
|
||||
// NEVER match at `tailStart`, no matter what is appended later.
|
||||
//
|
||||
// Append-only growth means existing characters are immutable while new ones
|
||||
// may appear after them, so "closed" may only be concluded from a present
|
||||
// character that contradicts every possible continuation (e.g. tail "1x" can
|
||||
// never grow into an ordered item, but tail "1" can become "1. c"). Running
|
||||
// out of text mid-marker therefore answers "may continue".
|
||||
//
|
||||
// Returns true when the tail could still continue the list (or the list's
|
||||
// marker is unrecognizable) — the conservative "don't freeze" answer. marked
|
||||
// may break the list anyway when the matching line is also an hr (`- - -`);
|
||||
// treating that as "may continue" merely skips a freeze, never corrupts one.
|
||||
function listMayContinueAt(text: string, tailStart: number, listRaw: string): boolean {
|
||||
const marker = LIST_MARKER_RE.exec(listRaw);
|
||||
if (marker === null) return true; // unrecognized list shape — stay conservative
|
||||
const n = text.length;
|
||||
let i = tailStart;
|
||||
// `listItemRegex` allows up to 3 leading spaces (the caller's next-char
|
||||
// guard rejects whitespace at the final cut, but mirror the rule exactly).
|
||||
while (i < n && i - tailStart < 3 && text.charCodeAt(i) === 0x20 /* space */) i++;
|
||||
if (i >= n) return true;
|
||||
const bullet = marker[1];
|
||||
if (bullet !== undefined) {
|
||||
if (text[i] !== bullet) return false; // wrong marker char — closed forever
|
||||
i++;
|
||||
} else {
|
||||
// Ordered: 1-9 digits, then the same `.`/`)` delimiter.
|
||||
let digits = 0;
|
||||
while (i < n && digits < 10) {
|
||||
const c = text.charCodeAt(i);
|
||||
if (c < 0x30 /* 0 */ || c > 0x39 /* 9 */) break;
|
||||
digits++;
|
||||
i++;
|
||||
}
|
||||
if (digits === 0 || digits > 9) return false; // no digit run / too long — closed forever
|
||||
if (i >= n) return true; // delimiter (or more digits) may still arrive
|
||||
if (text[i] !== marker[2]) return false; // wrong delimiter — closed forever
|
||||
i++;
|
||||
}
|
||||
// After the marker: `(?:[\t ][^\n]*)?(?:\n|$)` — tab/space + anything, a
|
||||
// bare newline, or end-of-input (which appends can still extend).
|
||||
if (i >= n) return true;
|
||||
const after = text.charCodeAt(i);
|
||||
return after === 0x20 /* space */ || after === 0x09 /* tab */ || after === 0x0a /* \n */;
|
||||
}
|
||||
|
||||
const NO_BLOCK_BOUNDARY = { end: 0, count: 0 } as const;
|
||||
|
||||
/**
|
||||
* Offset just past the last token in `tokens` that closes a block on a hard
|
||||
* `"\n\n"` break, together with the number of tokens up to and including it.
|
||||
* `count === 0` means the run holds no usable boundary.
|
||||
*
|
||||
* `base` is where `tokens[0]` starts inside `text`. A boundary qualifies only
|
||||
* when splitting there is invisible to the lexer, i.e. `lex(head) ++ lex(tail)
|
||||
* === lex(text)`:
|
||||
* - The break must sit inside `text`. At end-of-text the next character is
|
||||
* unknown (and, while streaming, may still arrive), so the cut is deferred.
|
||||
* - The next character must start real block content. Whitespace means the
|
||||
* block separator straddles the cut — e.g. a fence followed by
|
||||
* `"\n\n\n- list"` — and the two lexes desync.
|
||||
* - A preceding `list` must be provably closed: CommonMark lets a same-marker
|
||||
* item continue the list across the blank line, and marked merges both into
|
||||
* one renumbered loose list (`listMayContinueAt`).
|
||||
*/
|
||||
function stableBlockBoundary(text: string, base: number, tokens: Token[]): { end: number; count: number } {
|
||||
let pos = base;
|
||||
let end = 0;
|
||||
let count = 0;
|
||||
for (let i = 0; i < tokens.length; i++) {
|
||||
const raw = tokens[i].raw;
|
||||
const tokenEnd = pos + raw.length;
|
||||
if (raw.endsWith("\n\n")) {
|
||||
const prev = i > 0 ? tokens[i - 1] : undefined;
|
||||
if (prev === undefined || prev.type !== "list" || !listMayContinueAt(text, tokenEnd, prev.raw)) {
|
||||
end = tokenEnd;
|
||||
count = i + 1;
|
||||
}
|
||||
}
|
||||
pos = tokenEnd;
|
||||
}
|
||||
if (count === 0 || end >= text.length) return NO_BLOCK_BOUNDARY;
|
||||
const next = text.charCodeAt(end);
|
||||
if (next === 0x20 /* space */ || next === 0x0a /* \n */) return NO_BLOCK_BOUNDARY;
|
||||
return { end, count };
|
||||
}
|
||||
|
||||
// Bun's regex engine skips the start-anchor optimization for several of marked's
|
||||
// block rules — `hr`, `lheading`, `table` and `html` are `^`-anchored
|
||||
// alternations of quantified branches — so each failing `exec` rescans the whole
|
||||
// remaining source instead of stopping at offset 0. Lexing is then quadratic in
|
||||
// document length: an 800 KB message costs ~41 s under Bun where Node/V8 needs
|
||||
// ~60 ms, and it runs on the render path, freezing the UI. Bounded windows keep
|
||||
// every scan short and restore linear behavior (~0.7 s for that same message).
|
||||
const LEX_WINDOW_BYTES = 2 * 1024;
|
||||
// Under this size a single pass beats probing for window boundaries; the
|
||||
// crossover measured on pathological Markdown sits around 16 KB.
|
||||
const WINDOWED_LEX_MIN_BYTES = 16 * 1024;
|
||||
|
||||
/**
|
||||
* Lex `text` in bounded windows, producing the exact token stream
|
||||
* `markdownParser.lexer(text)` would.
|
||||
*
|
||||
* Window cuts come from marked itself: a throwaway BLOCK-ONLY probe lex of the
|
||||
* window reports its last stable block boundary ({@link stableBlockBoundary})
|
||||
* and only that confirmed segment is handed to the real lexer; a window
|
||||
* holding no boundary doubles until it finds one or reaches the end. Probes
|
||||
* never run inline tokenization (their inlineQueue is discarded) — a boundary
|
||||
* is a property of block structure alone, and probe inline passes were the
|
||||
* dominant cost of an earlier revision. Block tokenization runs per window
|
||||
* while inline tokenization is deferred to the end — mirroring `Lexer.lex` —
|
||||
* so a `[label]: dest` definition anywhere in the document still resolves for
|
||||
* every inline span.
|
||||
*
|
||||
* A boundary requires some top-level token whose raw ends in `"\n\n"`, so a
|
||||
* window that contains no blank line cannot cut: each round starts at the next
|
||||
* `"\n\n"` (skipping straight to the end when there is none — e.g. a tail
|
||||
* that is one long tight list) instead of probing sizes that cannot succeed.
|
||||
*/
|
||||
function lexWindowed(text: string): Token[] {
|
||||
const lexer = new Lexer(markdownParser.defaults);
|
||||
let offset = 0;
|
||||
while (offset < text.length) {
|
||||
let segment = "";
|
||||
const nextBlank = text.indexOf("\n\n", offset);
|
||||
if (nextBlank === -1) {
|
||||
segment = text.slice(offset);
|
||||
} else {
|
||||
const minSize = Math.max(LEX_WINDOW_BYTES, nextBlank + 2 - offset);
|
||||
for (let size = minSize; segment.length === 0; size *= 2) {
|
||||
if (offset + size >= text.length) {
|
||||
segment = text.slice(offset);
|
||||
break;
|
||||
}
|
||||
const probe = new Lexer(markdownParser.defaults);
|
||||
probe.blockTokens(text.slice(offset, offset + size), probe.tokens);
|
||||
const boundary = stableBlockBoundary(text, offset, probe.tokens);
|
||||
if (boundary.count > 0) segment = text.slice(offset, boundary.end);
|
||||
}
|
||||
}
|
||||
lexer.blockTokens(segment, lexer.tokens);
|
||||
offset += segment.length;
|
||||
}
|
||||
for (const queued of lexer.inlineQueue) lexer.inlineTokens(queued.src, queued.tokens);
|
||||
lexer.inlineQueue = [];
|
||||
return lexer.tokens;
|
||||
}
|
||||
|
||||
/** Lex a whole document, windowing anything large enough for the quadratic scan to bite. */
|
||||
function lexDocument(text: string): Token[] {
|
||||
// A CR shifts every `raw` span (marked normalizes CRLF before tokenizing), so
|
||||
// window offsets would address the wrong characters — lex those in one pass.
|
||||
if (text.length < WINDOWED_LEX_MIN_BYTES || text.includes("\r")) return markdownParser.lexer(text);
|
||||
return lexWindowed(text);
|
||||
}
|
||||
|
||||
/** Drop all L2 cache entries. Call on theme change to prevent stale styled output. */
|
||||
export function clearRenderCache(): void {
|
||||
renderCache.clear();
|
||||
@@ -1219,12 +1514,12 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ
|
||||
text.length > prefix.length &&
|
||||
text.startsWith(prefix)
|
||||
) {
|
||||
const tailTokens = markdownParser.lexer(text.slice(prefix.length));
|
||||
const tailTokens = lexDocument(text.slice(prefix.length));
|
||||
const tokens = [...prefixTokens, ...tailTokens];
|
||||
this.#freezeStablePrefix(text, tokens, { preserveExisting: true });
|
||||
return tokens;
|
||||
}
|
||||
const tokens = markdownParser.lexer(text);
|
||||
const tokens = lexDocument(text);
|
||||
if (canStream) {
|
||||
this.#freezeStablePrefix(text, tokens, { preserveExisting: false });
|
||||
} else {
|
||||
@@ -1241,36 +1536,11 @@ export class Markdown implements Component, NativeScrollbackCommittedRows, Nativ
|
||||
// reference definitions, so each token's `raw` is a verbatim slice of `text`
|
||||
// and the summed offsets address `text` exactly.
|
||||
#freezeStablePrefix(text: string, tokens: Token[], opts: { preserveExisting: boolean }): void {
|
||||
let pos = 0;
|
||||
let frozenEnd = 0;
|
||||
let frozenCount = 0;
|
||||
for (let i = 0; i < tokens.length; i++) {
|
||||
const raw = tokens[i].raw;
|
||||
const end = pos + raw.length;
|
||||
// A `space` token ending in "\n\n" closes the preceding block, but a
|
||||
// `list` before it can still be extended by a following same-marker
|
||||
// item across the blank line (CommonMark loose-list continuation),
|
||||
// which marked merges into one renumbered loose list. Freezing across
|
||||
// such a cut would keep the lists separate. Never freeze right after a
|
||||
// list — it stays in the re-lexed tail.
|
||||
if (raw.endsWith("\n\n") && tokens[i - 1]?.type !== "list") {
|
||||
frozenEnd = end;
|
||||
frozenCount = i + 1;
|
||||
}
|
||||
pos = end;
|
||||
}
|
||||
// Freeze only when the tail begins with real block content. If the next
|
||||
// char is whitespace (an extra blank line, or an indented continuation),
|
||||
// the block separator straddles the cut and lex(prefix)++lex(tail) would
|
||||
// desync from a full lex — e.g. a fence followed by "\n\n\n- list". When
|
||||
// frozenEnd is at end-of-text the next char is unknown, so defer.
|
||||
if (frozenCount > 0 && frozenEnd < text.length) {
|
||||
const next = text.charCodeAt(frozenEnd);
|
||||
if (next !== 0x20 /* space */ && next !== 0x0a /* \n */) {
|
||||
this.#streamPrefixText = text.slice(0, frozenEnd);
|
||||
this.#streamPrefixTokens = tokens.slice(0, frozenCount);
|
||||
return;
|
||||
}
|
||||
const frozen = stableBlockBoundary(text, 0, tokens);
|
||||
if (frozen.count > 0) {
|
||||
this.#streamPrefixText = text.slice(0, frozen.end);
|
||||
this.#streamPrefixTokens = tokens.slice(0, frozen.count);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!opts.preserveExisting) {
|
||||
|
||||
@@ -288,8 +288,17 @@ export class KeybindingsManager {
|
||||
matches(data: string, keybinding: Keybinding): boolean {
|
||||
const parsed = parseKey(data);
|
||||
if (parsed === undefined) return false;
|
||||
const matchKeys = this.#matchKeysById.get(keybinding);
|
||||
return matchKeys?.has(canonicalKeyId(parsed)) ?? false;
|
||||
return this.matchesCanonical(canonicalKeyId(parsed), keybinding);
|
||||
}
|
||||
|
||||
/**
|
||||
* Set-lookup variant of {@link matches} for hot input paths: the caller
|
||||
* parses `data` once (`parseKey` + `canonicalKeyId`) and probes many
|
||||
* bindings without re-parsing the raw sequence per probe.
|
||||
*/
|
||||
matchesCanonical(canonical: string | undefined, keybinding: Keybinding): boolean {
|
||||
if (canonical === undefined) return false;
|
||||
return this.#matchKeysById.get(keybinding)?.has(canonical) ?? false;
|
||||
}
|
||||
|
||||
getKeys(keybinding: Keybinding): KeyId[] {
|
||||
|
||||
@@ -286,7 +286,7 @@ export function emergencyTerminalRestore(): void {
|
||||
terminal.write("\x1b[?1049l");
|
||||
altScreenActive = false;
|
||||
}
|
||||
terminal.showCursor();
|
||||
terminal.showCursor(true);
|
||||
} else if (terminalEverStarted && !isTerminalHeadless()) {
|
||||
// Blind restore only if we know a terminal was started but lost track of it
|
||||
// This avoids writing escape sequences for non-TUI commands (grep, commit, etc.)
|
||||
@@ -362,9 +362,11 @@ export interface Terminal {
|
||||
// Cursor positioning (relative to current position)
|
||||
moveBy(lines: number): void; // Move cursor up (negative) or down (positive) by N lines
|
||||
|
||||
// Cursor visibility
|
||||
hideCursor(): void; // Hide the cursor
|
||||
showCursor(): void; // Show the cursor
|
||||
// Cursor visibility. Same-state calls are deduped against the visibility
|
||||
// last written to the terminal; pass force=true to write unconditionally
|
||||
// (crash/exit restore paths).
|
||||
hideCursor(force?: boolean): void; // Hide the cursor
|
||||
showCursor(force?: boolean): void; // Show the cursor
|
||||
|
||||
// Clear operations
|
||||
clearLine(): void; // Clear current line
|
||||
@@ -478,6 +480,12 @@ export class ProcessTerminal implements Terminal {
|
||||
this.#markTerminalDisconnected("stdin failed", err);
|
||||
};
|
||||
#dead = false;
|
||||
// Last cursor visibility written to the terminal, sniffed from every
|
||||
// outgoing sequence (frame buffers embed their own ?25h/?25l), so
|
||||
// hideCursor()/showCursor() can skip same-state writes. `undefined` =
|
||||
// unknown (fresh start, resize, or an alt-screen switch newer than the
|
||||
// last cursor sequence — some hosts keep DECTCEM per buffer).
|
||||
#cursorVisible: boolean | undefined;
|
||||
// Captured at construction and re-read at start(): when true, every real
|
||||
// terminal side effect (writes, probes, raw mode, SIGWINCH, timers) is
|
||||
// suppressed. Defaults on under `bun test` — see isTerminalHeadless().
|
||||
@@ -575,6 +583,8 @@ export class ProcessTerminal implements Terminal {
|
||||
this.#inputHandler = onInput;
|
||||
this.#resizeHandler = onResize;
|
||||
this.#disconnectHandler = onDisconnect;
|
||||
// The host terminal's cursor visibility is unknown until we write it.
|
||||
this.#cursorVisible = undefined;
|
||||
|
||||
// Headless (tests): suppress every real-terminal side effect. Skip raw
|
||||
// mode, stdin listeners, capability probes, SIGWINCH, and emergency-restore
|
||||
@@ -620,6 +630,9 @@ export class ProcessTerminal implements Terminal {
|
||||
// dimensions before firing `resize`, so it is authoritative for geometry:
|
||||
// reconcile any stale cached DEC 2048 report before notifying the renderer.
|
||||
this.#stdoutResizeListener = () => {
|
||||
// Conservative: some hosts reset modes across a resize/reattach, so
|
||||
// re-establish cursor visibility on the next explicit call.
|
||||
this.#cursorVisible = undefined;
|
||||
this.#reconcileInBandGeometryOnResize();
|
||||
this.#resizeHandler?.();
|
||||
};
|
||||
@@ -1474,6 +1487,9 @@ export class ProcessTerminal implements Terminal {
|
||||
}
|
||||
this.#stdoutErrorCleanup?.();
|
||||
this.#stdoutErrorCleanup = undefined;
|
||||
// After stop() the terminal is shared with other writers; visibility
|
||||
// tracking is only meaningful while this instance owns the TTY.
|
||||
this.#cursorVisible = undefined;
|
||||
}
|
||||
|
||||
#ensureStdoutErrorHandler(): void {
|
||||
@@ -1527,6 +1543,7 @@ export class ProcessTerminal implements Terminal {
|
||||
// files). They serve no purpose there and would surface as visible noise.
|
||||
if (!process.stdout.isTTY) return;
|
||||
this.#ensureStdoutErrorHandler();
|
||||
this.#trackCursorVisibility(data);
|
||||
// A console-sharing child process may have flipped the console codepage
|
||||
// away from UTF-8; repair it before any bytes hit WriteFile so no frame
|
||||
// is ever translated through an OEM codepage. See ensureWindowsConsoleUtf8.
|
||||
@@ -1578,14 +1595,37 @@ export class ProcessTerminal implements Terminal {
|
||||
// lines === 0: no movement
|
||||
}
|
||||
|
||||
hideCursor(): void {
|
||||
hideCursor(force = false): void {
|
||||
if (!force && this.#cursorVisible === false) return;
|
||||
this.#safeWrite("\x1b[?25l");
|
||||
}
|
||||
|
||||
showCursor(): void {
|
||||
showCursor(force = false): void {
|
||||
if (!force && this.#cursorVisible === true) return;
|
||||
this.#safeWrite("\x1b[?25h");
|
||||
}
|
||||
|
||||
/**
|
||||
* Sniff outgoing data for the last cursor-visibility change so the tracked
|
||||
* state stays correct for sequences embedded in frame buffers
|
||||
* (TUI#cursorControlSequence appends ?25h/?25l inside the paint write). An
|
||||
* alt-screen switch (DECSET/DECRST 1049) newer than the last cursor
|
||||
* sequence resets tracking to unknown: some hosts keep DECTCEM per buffer.
|
||||
*/
|
||||
#trackCursorVisibility(data: string): void {
|
||||
let idx = data.lastIndexOf("\x1b[?25");
|
||||
while (idx !== -1) {
|
||||
const final = data.charCodeAt(idx + 5);
|
||||
if (final === 0x68 /* h */ || final === 0x6c /* l */) break;
|
||||
idx = idx === 0 ? -1 : data.lastIndexOf("\x1b[?25", idx - 1);
|
||||
}
|
||||
if (data.lastIndexOf("\x1b[?1049") > idx) {
|
||||
this.#cursorVisible = undefined;
|
||||
return;
|
||||
}
|
||||
if (idx !== -1) this.#cursorVisible = data.charCodeAt(idx + 5) === 0x68;
|
||||
}
|
||||
|
||||
clearLine(): void {
|
||||
this.#safeWrite("\x1b[K");
|
||||
}
|
||||
|
||||
@@ -1875,7 +1875,9 @@ export class TUI extends Container {
|
||||
this.terminal.write(targetRow <= viewportBottom ? "\r" : "\r\n");
|
||||
}
|
||||
|
||||
this.terminal.showCursor();
|
||||
// Force: the parent shell needs the cursor back regardless of what the
|
||||
// terminal-level dedupe believes was last written.
|
||||
this.terminal.showCursor(true);
|
||||
this.#forgetHardwareCursorState();
|
||||
this.terminal.stop();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import { ProcessTerminal } from "@oh-my-pi/pi-tui/terminal";
|
||||
import { setTerminalHeadless } from "@oh-my-pi/pi-utils";
|
||||
|
||||
// ProcessTerminal dedupes cursor-visibility writes: hideCursor()/showCursor()
|
||||
// skip the ?25l/?25h escape when the terminal already holds that state. The
|
||||
// tracked state is sniffed from every outgoing write, so cursor sequences
|
||||
// embedded in frame buffers (TUI appends ?25h/?25l inside the paint write)
|
||||
// keep it in sync, and an alt-screen switch resets it to unknown. Crash/exit
|
||||
// restore paths pass force=true and must always write.
|
||||
|
||||
const HIDE = "\x1b[?25l";
|
||||
const SHOW = "\x1b[?25h";
|
||||
|
||||
const stdinIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "isTTY");
|
||||
const stdoutIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "isTTY");
|
||||
const stdinSetRawModeDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "setRawMode");
|
||||
let previousHeadless = false;
|
||||
|
||||
function restoreProperty(target: object, key: string, descriptor: PropertyDescriptor | undefined): void {
|
||||
if (descriptor) {
|
||||
Object.defineProperty(target, key, descriptor);
|
||||
return;
|
||||
}
|
||||
delete (target as Record<string, unknown>)[key];
|
||||
}
|
||||
|
||||
function startCapturedTerminal() {
|
||||
const writes: string[] = [];
|
||||
Object.defineProperty(process.stdin, "isTTY", { value: true, configurable: true });
|
||||
Object.defineProperty(process.stdout, "isTTY", { value: true, configurable: true });
|
||||
Object.defineProperty(process.stdin, "setRawMode", { value: vi.fn(), configurable: true });
|
||||
vi.spyOn(process, "kill").mockReturnValue(true);
|
||||
vi.spyOn(process.stdin, "resume").mockImplementation(() => process.stdin);
|
||||
vi.spyOn(process.stdin, "pause").mockImplementation(() => process.stdin);
|
||||
vi.spyOn(process.stdin, "setEncoding").mockImplementation(() => process.stdin);
|
||||
vi.spyOn(process.stdout, "write").mockImplementation(chunk => {
|
||||
writes.push(String(chunk));
|
||||
return true;
|
||||
});
|
||||
|
||||
const terminal = new ProcessTerminal();
|
||||
terminal.start(
|
||||
() => {},
|
||||
() => {},
|
||||
);
|
||||
writes.length = 0;
|
||||
return { terminal, writes };
|
||||
}
|
||||
|
||||
describe("ProcessTerminal cursor-visibility dedupe", () => {
|
||||
beforeEach(() => {
|
||||
previousHeadless = setTerminalHeadless(false);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
setTerminalHeadless(previousHeadless);
|
||||
vi.restoreAllMocks();
|
||||
restoreProperty(process.stdin, "isTTY", stdinIsTtyDescriptor);
|
||||
restoreProperty(process.stdout, "isTTY", stdoutIsTtyDescriptor);
|
||||
restoreProperty(process.stdin, "setRawMode", stdinSetRawModeDescriptor);
|
||||
});
|
||||
|
||||
it("writes each visibility change once and skips same-state repeats", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
terminal.hideCursor();
|
||||
terminal.hideCursor();
|
||||
terminal.hideCursor();
|
||||
expect(writes).toEqual([HIDE]);
|
||||
|
||||
terminal.showCursor();
|
||||
terminal.showCursor();
|
||||
expect(writes).toEqual([HIDE, SHOW]);
|
||||
|
||||
terminal.hideCursor();
|
||||
expect(writes).toEqual([HIDE, SHOW, HIDE]);
|
||||
terminal.stop();
|
||||
});
|
||||
|
||||
it("tracks cursor sequences embedded in frame writes", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
// A paint that repositions the hardware cursor ends by showing it.
|
||||
terminal.write(`\x1b[2Bframe content\x1b[5G${SHOW}\x1b[?2026l`);
|
||||
writes.length = 0;
|
||||
|
||||
terminal.showCursor(); // already visible per the frame write
|
||||
expect(writes).toEqual([]);
|
||||
|
||||
terminal.hideCursor(); // state change: must write
|
||||
expect(writes).toEqual([HIDE]);
|
||||
terminal.stop();
|
||||
});
|
||||
|
||||
it("honors the last of multiple cursor sequences in one write", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
terminal.write(`${SHOW}overlay paint${HIDE}`);
|
||||
writes.length = 0;
|
||||
|
||||
terminal.hideCursor();
|
||||
expect(writes).toEqual([]);
|
||||
terminal.showCursor();
|
||||
expect(writes).toEqual([SHOW]);
|
||||
terminal.stop();
|
||||
});
|
||||
|
||||
it("force-writes regardless of tracked state (crash/exit restore contract)", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
terminal.showCursor();
|
||||
terminal.showCursor(true);
|
||||
terminal.showCursor(true);
|
||||
expect(writes).toEqual([SHOW, SHOW, SHOW]);
|
||||
|
||||
terminal.hideCursor();
|
||||
terminal.hideCursor(true);
|
||||
expect(writes).toEqual([SHOW, SHOW, SHOW, HIDE, HIDE]);
|
||||
terminal.stop();
|
||||
});
|
||||
|
||||
it("resets tracking to unknown when an alt-screen switch follows the last cursor sequence", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
terminal.hideCursor();
|
||||
// Alt-screen enter after the hide: some hosts keep DECTCEM per buffer,
|
||||
// so the tracked state is no longer trustworthy.
|
||||
terminal.write("\x1b[?1049h\x1b[2J");
|
||||
writes.length = 0;
|
||||
|
||||
terminal.hideCursor();
|
||||
expect(writes).toEqual([HIDE]);
|
||||
terminal.stop();
|
||||
});
|
||||
|
||||
it("does not confuse other private modes with cursor visibility", () => {
|
||||
const { terminal, writes } = startCapturedTerminal();
|
||||
|
||||
terminal.hideCursor();
|
||||
writes.length = 0;
|
||||
// Neither DECRQM on mode 25 nor unrelated ?25xx modes change visibility.
|
||||
terminal.write("\x1b[?25$p\x1b[?2026h");
|
||||
terminal.hideCursor();
|
||||
expect(writes).toEqual(["\x1b[?25$p\x1b[?2026h"]);
|
||||
terminal.stop();
|
||||
});
|
||||
});
|
||||
@@ -2396,6 +2396,110 @@ describe("Editor component", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("Bulk input fast path and paste iteration", () => {
|
||||
it("produces identical state for a chunked paste and a single-sequence paste", () => {
|
||||
const content = "alpha beta\ngamma delta\nepsilon";
|
||||
const single = new Editor(defaultEditorTheme);
|
||||
single.handleInput(`\x1b[200~${content}\x1b[201~`);
|
||||
|
||||
const chunked = new Editor(defaultEditorTheme);
|
||||
chunked.handleInput("\x1b[200~");
|
||||
for (const ch of content) chunked.handleInput(ch);
|
||||
chunked.handleInput("\x1b[201~");
|
||||
|
||||
expect(chunked.getText()).toBe(single.getText());
|
||||
expect(chunked.getCursor()).toEqual(single.getCursor());
|
||||
});
|
||||
|
||||
it("normalizes CRLF identically for single and chunked paste delivery", () => {
|
||||
const single = new Editor(defaultEditorTheme);
|
||||
single.handleInput("\x1b[200~one\r\ntwo\rthree\x1b[201~");
|
||||
|
||||
const chunked = new Editor(defaultEditorTheme);
|
||||
chunked.handleInput("\x1b[200~one\r");
|
||||
chunked.handleInput("\ntwo");
|
||||
chunked.handleInput("\rthree\x1b[201~");
|
||||
|
||||
expect(single.getText()).toBe("one\ntwo\nthree");
|
||||
expect(chunked.getText()).toBe(single.getText());
|
||||
expect(chunked.getCursor()).toEqual(single.getCursor());
|
||||
});
|
||||
|
||||
it("processes paste remainders iteratively, applying every trailing paste and keystroke", () => {
|
||||
const editor = new Editor(defaultEditorTheme);
|
||||
// One read carrying two complete pastes plus trailing typed text: the
|
||||
// remainder after each paste loops back through input handling.
|
||||
editor.handleInput("\x1b[200~ab\x1b[201~\x1b[200~cd\x1b[201~ef");
|
||||
expect(editor.getText()).toBe("abcdef");
|
||||
expect(editor.getCursor()).toEqual({ line: 0, col: 6 });
|
||||
});
|
||||
|
||||
it("handles a long train of pastes in one read without recursing per remainder", () => {
|
||||
const editor = new Editor(defaultEditorTheme);
|
||||
editor.handleInput("\x1b[200~x\x1b[201~".repeat(2000));
|
||||
expect(editor.getText()).toBe("x".repeat(2000));
|
||||
});
|
||||
|
||||
it("inserts a plain printable run identically to per-scalar delivery", () => {
|
||||
const run = "The quick brown fox 123 -_. naïve 😀 path";
|
||||
const bulk = new Editor(defaultEditorTheme);
|
||||
bulk.handleInput(run);
|
||||
|
||||
const perChar = new Editor(defaultEditorTheme);
|
||||
for (const ch of run) perChar.handleInput(ch);
|
||||
|
||||
expect(bulk.getText()).toBe(perChar.getText());
|
||||
expect(bulk.getCursor()).toEqual(perChar.getCursor());
|
||||
});
|
||||
|
||||
it("keeps escape sequences interleaved with printable runs on the dispatch path", () => {
|
||||
const editor = new Editor(defaultEditorTheme);
|
||||
editor.handleInput("abc");
|
||||
editor.handleInput("\x1b[D"); // Left
|
||||
editor.handleInput("XY"); // bulk run lands before "c"
|
||||
expect(editor.getText()).toBe("abXYc");
|
||||
expect(editor.getCursor()).toEqual({ line: 0, col: 4 });
|
||||
});
|
||||
|
||||
it("opens @ autocomplete when the trigger arrives inside a bulk printable run", async () => {
|
||||
const editor = new Editor(defaultEditorTheme);
|
||||
const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers<void>();
|
||||
editor.setAutocompleteProvider({
|
||||
async getSuggestions() {
|
||||
return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" };
|
||||
},
|
||||
applyCompletion(lines, cursorLine, cursorCol) {
|
||||
return { lines, cursorLine, cursorCol };
|
||||
},
|
||||
});
|
||||
editor.onAutocompleteUpdate = resolveAutocompleteUpdated;
|
||||
|
||||
editor.handleInput("see @sr"); // one bulk run ending in an @-token
|
||||
|
||||
await autocompleteUpdated;
|
||||
expect(editor.isShowingAutocomplete()).toBe(true);
|
||||
});
|
||||
|
||||
it("opens @ autocomplete after a bracketed paste ending in a trigger token", async () => {
|
||||
const editor = new Editor(defaultEditorTheme);
|
||||
const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers<void>();
|
||||
editor.setAutocompleteProvider({
|
||||
async getSuggestions() {
|
||||
return { items: [{ label: "src/", value: "src/" }], prefix: "@sr" };
|
||||
},
|
||||
applyCompletion(lines, cursorLine, cursorCol) {
|
||||
return { lines, cursorLine, cursorCol };
|
||||
},
|
||||
});
|
||||
editor.onAutocompleteUpdate = resolveAutocompleteUpdated;
|
||||
|
||||
editor.handleInput("\x1b[200~see @sr\x1b[201~");
|
||||
|
||||
await autocompleteUpdated;
|
||||
expect(editor.isShowingAutocomplete()).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("Korean NFC paste normalization", () => {
|
||||
// macOS Finder drag-drops/Copy-As-Pathname emit Korean filenames as
|
||||
// NFD (decomposed) — e.g. `화` becomes `ᄒ`(U+1112) + `ᅪ`(U+116A).
|
||||
|
||||
@@ -243,4 +243,87 @@ describe("Markdown incremental streaming lex (E2)", () => {
|
||||
expect(streamLines).toEqual(renderCold(crlf.slice(0, len), 60));
|
||||
}
|
||||
});
|
||||
|
||||
// Closed-list lookahead: a "\n\n" boundary directly after a list token is
|
||||
// freezable iff the tail cannot start a continuation item of that list
|
||||
// (same bullet char, or 1-9 digits + same delimiter — marked's
|
||||
// listItemRegex). These corpora cross list/non-list and
|
||||
// list/incompatible-list boundaries; the divergence (and the freeze
|
||||
// opportunity) is phase-sensitive, so each runs at step=1 and the
|
||||
// production reveal granularity (step=3).
|
||||
it("bullet list followed by a paragraph grows byte-identically", () => {
|
||||
const doc =
|
||||
"- alpha item with words\n- beta item with words\n- gamma item\n\n" +
|
||||
"Closing paragraph that keeps streaming additional words to the end.";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
assertIdenticalGrowthTransient(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("bullet list followed by a different-marker list stays two lists", () => {
|
||||
const doc = "- alpha\n- beta\n\n* starred one\n* starred two\n\n+ plus one\n+ plus two";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("ordered list followed by a paren-delimited list stays two lists", () => {
|
||||
const doc = "1. dot one\n2. dot two\n\n1) paren one\n2) paren two";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
assertIdenticalGrowthTransient(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("list followed by blockquote grows byte-identically", () => {
|
||||
const doc = "- alpha\n- beta\n\n> quoted line one with words\n> quoted line two here";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("list followed by heading grows byte-identically", () => {
|
||||
const doc = "1. one\n2. two\n\n# Heading after the list\n\nTail prose keeps going on.";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("list followed by fenced code grows byte-identically", () => {
|
||||
const doc = "- alpha\n- beta\n\n```ts\nconst x = compute(a, b);\nreturn x;\n```\n\ntail text";
|
||||
assertIdenticalGrowth(doc, 60, 1);
|
||||
assertIdenticalGrowth(doc, 60, 3);
|
||||
});
|
||||
|
||||
it("a same-marker list across a blank line still merges while growing", () => {
|
||||
const bullets = "- a\n- b\n\n- c\n- d";
|
||||
assertIdenticalGrowth(bullets, 60, 1);
|
||||
assertIdenticalGrowth(bullets, 60, 3);
|
||||
});
|
||||
|
||||
it("a list closed by a paragraph freezes at the boundary (streaming perf gate)", () => {
|
||||
// The lookahead must actually fire here: the tail after the blank line
|
||||
// is a paragraph, which cannot continue a `-` list, so the rendered
|
||||
// list rows become settled (frozen prefix) on the transient path.
|
||||
const doc = "- alpha\n- beta\n- gamma\n\nClosing paragraph after the list keeps going.";
|
||||
const streaming = new Markdown("", 0, 0, THEME);
|
||||
streaming.transientRenderCache = true;
|
||||
clearRenderCache();
|
||||
streaming.setText(doc);
|
||||
const streamLines = streaming.render(60);
|
||||
expect(streamLines).toEqual(renderCold(doc, 60));
|
||||
expect(streaming.getLastRenderSettledRows()).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("a document that is one still-growing list never freezes mid-list", () => {
|
||||
// No (b)-style intra-list freezing shipped: loose/tight and ordered
|
||||
// renumbering are whole-list properties, so no prefix of an open list
|
||||
// is byte-stable. Settled rows must stay 0 for a pure-list document.
|
||||
const doc = "- one two three\n- four five six\n\n- seven eight nine";
|
||||
const streaming = new Markdown("", 0, 0, THEME);
|
||||
streaming.transientRenderCache = true;
|
||||
for (let len = 1; len <= doc.length; len += 1) {
|
||||
clearRenderCache();
|
||||
streaming.setText(doc.slice(0, len));
|
||||
const streamLines = streaming.render(60);
|
||||
expect(streamLines).toEqual(renderCold(doc.slice(0, len), 60));
|
||||
expect(streaming.getLastRenderSettledRows()).toBe(0);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,6 +1,13 @@
|
||||
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
|
||||
import { stripVTControlCharacters } from "node:util";
|
||||
import { clearRenderCache, Markdown, renderInlineMarkdown } from "@oh-my-pi/pi-tui/components/markdown";
|
||||
import {
|
||||
autolinkSchemeScanIndex,
|
||||
clearRenderCache,
|
||||
Markdown,
|
||||
mathStartIndex,
|
||||
renderInlineMarkdown,
|
||||
urlTokenPossible,
|
||||
} from "@oh-my-pi/pi-tui/components/markdown";
|
||||
import { setTerminalTextSizing, TERMINAL } from "@oh-my-pi/pi-tui/terminal-capabilities";
|
||||
import { type Component, TUI } from "@oh-my-pi/pi-tui/tui";
|
||||
import { visibleWidth } from "@oh-my-pi/pi-tui/utils";
|
||||
@@ -1940,9 +1947,11 @@ describe("Markdown.render reference stability", () => {
|
||||
});
|
||||
|
||||
it("does not share oversized renders through the L2 cache", () => {
|
||||
// Fixture must exceed RENDER_CACHE_MAX_ENTRY_SIZE (256 KiB of rendered
|
||||
// lines) so the entry is rejected and each render owns its array.
|
||||
const width = 80;
|
||||
const paragraph = `cache-budget sentinel ${"x".repeat(120)}`;
|
||||
const largeText = Array.from({ length: 160 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n");
|
||||
const largeText = Array.from({ length: 1400 }, (_, index) => `Paragraph ${index}: ${paragraph}`).join("\n\n");
|
||||
|
||||
const first = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width);
|
||||
const second = new Markdown(largeText, 0, 0, defaultMarkdownTheme).render(width);
|
||||
@@ -2319,3 +2328,146 @@ describe("Math rendering", () => {
|
||||
expect(lines[fxIdx + 1]).toContain("x > 0");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline start()/url-gate scanners (perf rewrites)", () => {
|
||||
// The hand-rolled scanners replaced regex scans that marked runs on the
|
||||
// remaining source at every inline position. They must return exactly what
|
||||
// the old regexes returned for every input.
|
||||
const OLD_MATH_START = /\$|\\\(|\\\[/;
|
||||
const OLD_AUTOLINK_SCAN = /www\.|https?:\/\/|ftp:\/\//i;
|
||||
// marked's bundled GFM inline url rule (verbatim, no flags).
|
||||
const GFM_URL_REGEX =
|
||||
/^((?:[hH][tT][tT][pP][sS]?|[fF][tT][pP]):\/\/|www\.)(?:[a-zA-Z0-9-]+\.?)+[^\s<]*|^[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/;
|
||||
|
||||
const fixtures = [
|
||||
"",
|
||||
"plain prose with no candidates at all",
|
||||
"$x$ math first",
|
||||
"prose then $inline$ math",
|
||||
"prose then \\(paren\\) math",
|
||||
"prose then \\[bracket\\] math",
|
||||
"\\( before $ dollar",
|
||||
"$ before \\( paren",
|
||||
"backslash only \\ then ( apart",
|
||||
"ends with backslash \\",
|
||||
"ends with dollar $",
|
||||
"www.example.com leading",
|
||||
"see www.example.com mid-string",
|
||||
"see WWW.EXAMPLE.COM upper",
|
||||
"mixed WwW.case.com scan",
|
||||
"http://example.com leading",
|
||||
"prose http://example.com mid",
|
||||
"prose HTTPS://EXAMPLE.COM upper",
|
||||
"HtTpS://mixed.example",
|
||||
"ftp://files.example mid ftp",
|
||||
"prose FTP://FILES.EXAMPLE",
|
||||
"ftps:// is not ftp:// until here ftp://x",
|
||||
"wwww.overlap.example",
|
||||
"hhttp://overlap.example",
|
||||
"http:/ missing slash then https://real.example",
|
||||
"www without dot www. with dot",
|
||||
"w h f teaser chars but no scheme",
|
||||
"user@example.com email",
|
||||
"prose user.name+tag@example.co.uk",
|
||||
"trailing at sign only@ ",
|
||||
"@leading-at no local part",
|
||||
"a".repeat(400), // long identifier run, no @
|
||||
`${"a".repeat(400)}@example.com`, // long local part (past gate scan limit)
|
||||
"short@x",
|
||||
"dots...and+plus_under-score@host.tld",
|
||||
];
|
||||
|
||||
it("mathStartIndex matches the old /\\$|\\\\\\(|\\\\\\[/ scan on every fixture", () => {
|
||||
for (const src of fixtures) {
|
||||
const m = OLD_MATH_START.exec(src);
|
||||
expect(mathStartIndex(src)).toBe(m ? m.index : undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("autolinkSchemeScanIndex matches the old /www\\.|https?:\\/\\/|ftp:\\/\\//i scan on every fixture", () => {
|
||||
for (const src of fixtures) {
|
||||
const m = OLD_AUTOLINK_SCAN.exec(src);
|
||||
expect(autolinkSchemeScanIndex(src)).toBe(m ? m.index : undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("urlTokenPossible is conservative: never false when the GFM url regex matches", () => {
|
||||
for (const src of fixtures) {
|
||||
if (GFM_URL_REGEX.test(src)) {
|
||||
expect(urlTokenPossible(src)).toBeTrue();
|
||||
}
|
||||
}
|
||||
// And it actually gates: plain prose with no scheme/email head is rejected.
|
||||
expect(urlTokenPossible("plain prose, nothing linkable here")).toBeFalse();
|
||||
expect(urlTokenPossible("@leading-at no local part")).toBeFalse();
|
||||
});
|
||||
|
||||
it("gated tokenizer still autolinks urls and emails end-to-end", () => {
|
||||
const rendered = renderInlineMarkdown("see https://example.com and mail user@example.com now", {
|
||||
...defaultMarkdownTheme,
|
||||
link: (text: string) => `<L>${text}</L>`,
|
||||
});
|
||||
const plain = stripVTControlCharacters(rendered);
|
||||
expect(plain).toContain("<L>https://example.com</L>");
|
||||
expect(plain).toContain("<L>user@example.com</L>");
|
||||
});
|
||||
});
|
||||
|
||||
describe("windowed lexing (documents past WINDOWED_LEX_MIN_BYTES)", () => {
|
||||
// Large documents are lexed in bounded windows because Bun's regex engine
|
||||
// rescans the whole remaining source for marked's `^`-anchored block rules.
|
||||
// Every construct below straddles window cuts; a bad cut is visible in the
|
||||
// rendered output.
|
||||
afterEach(() => clearRenderCache());
|
||||
|
||||
const filler = (label: string, lines: number) =>
|
||||
Array.from({ length: lines }, (_, i) => `${label} paragraph ${i} with enough prose to fill a window.`).join(
|
||||
"\n\n",
|
||||
);
|
||||
|
||||
const plain = (text: string, width = 100) =>
|
||||
new Markdown(text, 0, 0, defaultMarkdownTheme)
|
||||
.render(width)
|
||||
.map(line => stripVTControlCharacters(line).trimEnd());
|
||||
|
||||
it("resolves a reference definition that lands in a later window", () => {
|
||||
const doc = `Follow [the label][ref] first.\n\n${filler("body", 400)}\n\n[ref]: https://example.com/late\n`;
|
||||
expect(doc.length).toBeGreaterThan(16 * 1024);
|
||||
|
||||
const rendered = plain(doc, 120);
|
||||
// The reflink resolved: marked emitted a link token (rendered as
|
||||
// `label (href)`), so the raw `[label][ref]` syntax is gone and the
|
||||
// definition line itself produced no output block of its own.
|
||||
expect(rendered[0]).toBe("Follow the label (https://example.com/late) first.");
|
||||
expect(rendered.filter(line => line.includes("https://example.com/late"))).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("keeps a fenced block longer than one window intact", () => {
|
||||
const code = Array.from({ length: 200 }, (_, i) => `const value${i} = ${i};`).join("\n");
|
||||
const doc = `${filler("intro", 300)}\n\n\`\`\`ts\n${code}\n\`\`\`\n\n${filler("outro", 20)}`;
|
||||
expect(code.length).toBeGreaterThan(2 * 1024);
|
||||
|
||||
const rendered = plain(doc);
|
||||
// Exactly one fence pair: a window cut inside the block would close and
|
||||
// reopen it (or spill code lines into prose).
|
||||
expect(rendered.filter(line => line.trimStart().startsWith("```"))).toHaveLength(2);
|
||||
const first = rendered.findIndex(line => line.includes("const value0 = 0;"));
|
||||
expect(first).toBeGreaterThan(-1);
|
||||
for (let i = 0; i < 200; i++) {
|
||||
expect(rendered[first + i]).toContain(`const value${i} = ${i};`);
|
||||
}
|
||||
});
|
||||
|
||||
it("numbers an ordered list continuously across window cuts", () => {
|
||||
const items = Array.from({ length: 400 }, (_, i) => `${i + 1}. item ${i} padded with extra words to add bytes`);
|
||||
const doc = `${filler("intro", 60)}\n\n${items.join("\n")}\n`;
|
||||
expect(doc.length).toBeGreaterThan(16 * 1024);
|
||||
|
||||
const rendered = plain(doc, 120);
|
||||
for (const n of [1, 137, 400]) {
|
||||
expect(rendered.some(line => line.includes(`${n}. item ${n - 1} `))).toBe(true);
|
||||
}
|
||||
// A window cut that restarted the list would renumber later items.
|
||||
expect(rendered.filter(line => line.includes(" 1. item 0 ")).length).toBeLessThanOrEqual(1);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user