From 571eb5c4db7d93b908e1c14e7936b50c80960845 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 20 Jun 2026 07:38:58 +0200 Subject: [PATCH] fix(tui): cleaned up markdown rendering of inline tags and entities - Stripped `` and `` tags from rendered markdown while preserving their contents. - Updated HTML entity normalization to handle common special characters and numeric codes within the TUI. - Ensured whitespace surrounding stripped inline tags is preserved, preventing content merging. --- .../src/components/transcript/Markdown.tsx | 40 +++++++++++++- packages/collab-web/test/markdown.test.tsx | 14 +++++ packages/tui/CHANGELOG.md | 3 ++ packages/tui/src/components/markdown.ts | 54 ++++++++++++++++--- packages/tui/test/markdown.test.ts | 35 ++++++++++++ 5 files changed, 139 insertions(+), 7 deletions(-) diff --git a/packages/collab-web/src/components/transcript/Markdown.tsx b/packages/collab-web/src/components/transcript/Markdown.tsx index 33dcb5759..fcac4046f 100644 --- a/packages/collab-web/src/components/transcript/Markdown.tsx +++ b/packages/collab-web/src/components/transcript/Markdown.tsx @@ -10,7 +10,43 @@ function escapeHtml(s: string): string { .replaceAll('"', """) .replaceAll("'", "'"); } +function unescapeHtml(raw: string): string { + const parseCodePoint = (value: number): string => { + if (Number.isFinite(value) && value >= 0 && value <= 0x10ffff) { + try { + return String.fromCodePoint(value); + } catch (_) {} + } + return ""; + }; + return raw.replace(/&(amp|lt|gt|quot|apos|nbsp|#\d+|#x[0-9a-fA-F]+);/gi, (match, entity) => { + const lower = entity.toLowerCase(); + switch (lower) { + case "nbsp": + return " "; + case "lt": + return "<"; + case "gt": + return ">"; + case "quot": + return '"'; + case "apos": + return "'"; + case "amp": + return "&"; + default: { + if (lower.startsWith("#x")) { + return parseCodePoint(Number.parseInt(lower.slice(2), 16)); + } + if (lower.startsWith("#")) { + return parseCodePoint(Number(lower.slice(1))); + } + return match; + } + } + }); +} function safeHref(href: string): string | null { const trimmed = href.trim(); if (/^(?:https?:|mailto:)/i.test(trimmed)) return trimmed; @@ -23,7 +59,9 @@ const md = new Marked({ renderer: { // Raw HTML tokens (block + inline both arrive here) are escaped, never emitted. html({ text }) { - return escapeHtml(text); + const cleaned = text.replace(/<\/?(?:span|text)\b(?:\s[^>]*)?\s*\/?>/gi, ""); + if (cleaned === "") return ""; + return escapeHtml(unescapeHtml(cleaned)); }, link({ href, title, tokens }) { const inner = this.parser.parseInline(tokens); diff --git a/packages/collab-web/test/markdown.test.tsx b/packages/collab-web/test/markdown.test.tsx index 2935fe4a3..2a153ad87 100644 --- a/packages/collab-web/test/markdown.test.tsx +++ b/packages/collab-web/test/markdown.test.tsx @@ -29,4 +29,18 @@ describe("Transcript Markdown", () => { expect(html).toContain("<img src=x onerror=alert(1)>"); expect(html).not.toContain(" { + const html = renderMarkdown("▃"); + + expect(html).toContain("▃"); + expect(html).not.toContain("<span>"); + expect(html).not.toContain("<text>"); + }); + + it("unescapes HTML entities inside span and text HTML tags safely", () => { + const html = renderMarkdown("<▃> & "test" 😀 😀"); + + expect(html).toContain("<▃> & "test" 😀 😀"); + }); }); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5b818ef45..87ae5b9b4 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,9 @@ ## [Unreleased] +### Fixed + +- Fixed Markdown component to strip inline `` and `` tags while preserving their contents and unescaping nested HTML entities (`<`, `>`, `"`, `'`, `&`), preventing raw LLM block/inline formatting residues from leaking into rendered TUI output. ## [16.1.7] - 2026-06-20 ### Fixed diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index ef5187972..5540dac76 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -32,7 +32,43 @@ function isOsc66Line(line: string): boolean { } function normalizeHtmlEntitiesForTerminal(raw: string): string { - return raw.replace(/ /gi, " "); + const parseCodePoint = (value: number): string => { + if (Number.isFinite(value) && value >= 0 && value <= 0x10ffff) { + try { + return String.fromCodePoint(value); + } catch (_) { + // Fallback to empty string or original if invalid codepoint + } + } + return ""; + }; + + return raw.replace(/&(amp|lt|gt|quot|apos|nbsp|#\d+|#x[0-9a-fA-F]+);/gi, (match, entity) => { + const lower = entity.toLowerCase(); + switch (lower) { + case "nbsp": + return " "; + case "lt": + return "<"; + case "gt": + return ">"; + case "quot": + return '"'; + case "apos": + return "'"; + case "amp": + return "&"; + default: { + if (lower.startsWith("#x")) { + return parseCodePoint(Number.parseInt(lower.slice(2), 16)); + } + if (lower.startsWith("#")) { + return parseCodePoint(Number(lower.slice(1))); + } + return match; + } + } + }); } interface HtmlListState { @@ -50,7 +86,7 @@ function createHtmlNormalizationState(): HtmlNormalizationState { return { lists: [], openItems: [], itemHasContent: [] }; } -const HTML_TAG_REGEX = /<\/?(?:br|p|ol|ul|li)\b(?:\s[^>]*)?\s*\/?>/gi; +const HTML_TAG_REGEX = /<\/?(?:br|p|ol|ul|li|span|text)\b(?:\s[^>]*)?\s*\/?>/gi; function htmlTagName(tag: string): string { const match = /^<\/?\s*([A-Za-z][A-Za-z0-9:-]*)/.exec(tag); @@ -96,22 +132,28 @@ function normalizeHtmlForTerminal(raw: string, state: HtmlNormalizationState = c const tag = match[0]; const index = match.index ?? 0; const textBeforeTag = normalizeHtmlEntitiesForTerminal(raw.slice(lastIndex, index)); + const name = htmlTagName(tag); + // Every tag handled here is block-level EXCEPT span and text. For block-level tags, // HTML formatting whitespace between block/list tags (e.g. the newlines and // indentation in pretty-printed `
    \n
  • …`) is not rendered content; // appending it literally would leak source indentation before bullets and - // blank rows between items. Every tag handled here is block-level, so a - // whitespace-only slice is always insignificant formatting and is dropped. - if (textBeforeTag.trim() !== "") { + // blank rows between items. A whitespace-only slice is always insignificant formatting + // and is dropped. But for inline tags like span and text, surrounding whitespace + // is significant and must NOT be dropped. + const isInlineTag = name === "span" || name === "text"; + if (isInlineTag || textBeforeTag.trim() !== "") { output += textBeforeTag; markCurrentHtmlItemContent(state, textBeforeTag); } lastIndex = index + tag.length; - const name = htmlTagName(tag); const isClosing = /^<\//.test(tag); const isSelfClosing = /\/\s*>$/.test(tag); switch (name) { + case "span": + case "text": + break; case "br": output = appendHtmlLineBreak(output, true); break; diff --git a/packages/tui/test/markdown.test.ts b/packages/tui/test/markdown.test.ts index ecb13a8c2..854440484 100644 --- a/packages/tui/test/markdown.test.ts +++ b/packages/tui/test/markdown.test.ts @@ -1216,6 +1216,41 @@ bar`, "Should render HTML in code blocks", ).toBeTruthy(); }); + + it("should strip inline span and text HTML tags but keep their contents", () => { + const markdown = new Markdown("▃", 0, 0, defaultMarkdownTheme); + + const lines = markdown.render(80); + const plainLines = lines.map(line => stripVTControlCharacters(line).trim()); + const joinedPlain = plainLines.join(""); + + expect(joinedPlain).toBe("▃"); + }); + + it("should preserve whitespace surrounding stripped inline HTML tags", () => { + const markdown = new Markdown("some inner text", 0, 0, defaultMarkdownTheme); + + const lines = markdown.render(80); + const plainLines = lines.map(line => stripVTControlCharacters(line).trim()); + const joinedPlain = plainLines.join(""); + + expect(joinedPlain).toBe("some inner text"); + }); + + it("should unescape HTML entities inside and outside HTML tags", () => { + const markdown = new Markdown( + "<▃> & "test" 😀 😀", + 0, + 0, + defaultMarkdownTheme, + ); + + const lines = markdown.render(80); + const plainLines = lines.map(line => stripVTControlCharacters(line).trim()); + const joinedPlain = plainLines.join(""); + + expect(joinedPlain).toBe('<▃> & "test" 😀 😀'); + }); }); });