diff --git a/packages/collab-web/src/components/transcript/Markdown.tsx b/packages/collab-web/src/components/transcript/Markdown.tsx index 33dcb5759..fcac4046f 100644 --- a/packages/collab-web/src/components/transcript/Markdown.tsx +++ b/packages/collab-web/src/components/transcript/Markdown.tsx @@ -10,7 +10,43 @@ function escapeHtml(s: string): string { .replaceAll('"', """) .replaceAll("'", "'"); } +function unescapeHtml(raw: string): string { + const parseCodePoint = (value: number): string => { + if (Number.isFinite(value) && value >= 0 && value <= 0x10ffff) { + try { + return String.fromCodePoint(value); + } catch (_) {} + } + return ""; + }; + return raw.replace(/&(amp|lt|gt|quot|apos|nbsp|#\d+|#x[0-9a-fA-F]+);/gi, (match, entity) => { + const lower = entity.toLowerCase(); + switch (lower) { + case "nbsp": + return " "; + case "lt": + return "<"; + case "gt": + return ">"; + case "quot": + return '"'; + case "apos": + return "'"; + case "amp": + return "&"; + default: { + if (lower.startsWith("#x")) { + return parseCodePoint(Number.parseInt(lower.slice(2), 16)); + } + if (lower.startsWith("#")) { + return parseCodePoint(Number(lower.slice(1))); + } + return match; + } + } + }); +} function safeHref(href: string): string | null { const trimmed = href.trim(); if (/^(?:https?:|mailto:)/i.test(trimmed)) return trimmed; @@ -23,7 +59,9 @@ const md = new Marked({ renderer: { // Raw HTML tokens (block + inline both arrive here) are escaped, never emitted. html({ text }) { - return escapeHtml(text); + const cleaned = text.replace(/<\/?(?:span|text)\b(?:\s[^>]*)?\s*\/?>/gi, ""); + if (cleaned === "") return ""; + return escapeHtml(unescapeHtml(cleaned)); }, link({ href, title, tokens }) { const inner = this.parser.parseInline(tokens); diff --git a/packages/collab-web/test/markdown.test.tsx b/packages/collab-web/test/markdown.test.tsx index 2935fe4a3..2a153ad87 100644 --- a/packages/collab-web/test/markdown.test.tsx +++ b/packages/collab-web/test/markdown.test.tsx @@ -29,4 +29,18 @@ describe("Transcript Markdown", () => { expect(html).toContain("<img src=x onerror=alert(1)>"); expect(html).not.toContain(" { + const html = renderMarkdown("▃"); + + expect(html).toContain("▃"); + expect(html).not.toContain("<span>"); + expect(html).not.toContain("<text>"); + }); + + it("unescapes HTML entities inside span and text HTML tags safely", () => { + const html = renderMarkdown("<▃> & "test" 😀 😀"); + + expect(html).toContain("<▃> & "test" 😀 😀"); + }); }); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5b818ef45..87ae5b9b4 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,9 @@ ## [Unreleased] +### Fixed + +- Fixed Markdown component to strip inline `` and `` tags while preserving their contents and unescaping nested HTML entities (`<`, `>`, `"`, `'`, `&`), preventing raw LLM block/inline formatting residues from leaking into rendered TUI output. ## [16.1.7] - 2026-06-20 ### Fixed diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index ef5187972..5540dac76 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -32,7 +32,43 @@ function isOsc66Line(line: string): boolean { } function normalizeHtmlEntitiesForTerminal(raw: string): string { - return raw.replace(/ /gi, " "); + const parseCodePoint = (value: number): string => { + if (Number.isFinite(value) && value >= 0 && value <= 0x10ffff) { + try { + return String.fromCodePoint(value); + } catch (_) { + // Fallback to empty string or original if invalid codepoint + } + } + return ""; + }; + + return raw.replace(/&(amp|lt|gt|quot|apos|nbsp|#\d+|#x[0-9a-fA-F]+);/gi, (match, entity) => { + const lower = entity.toLowerCase(); + switch (lower) { + case "nbsp": + return " "; + case "lt": + return "<"; + case "gt": + return ">"; + case "quot": + return '"'; + case "apos": + return "'"; + case "amp": + return "&"; + default: { + if (lower.startsWith("#x")) { + return parseCodePoint(Number.parseInt(lower.slice(2), 16)); + } + if (lower.startsWith("#")) { + return parseCodePoint(Number(lower.slice(1))); + } + return match; + } + } + }); } interface HtmlListState { @@ -50,7 +86,7 @@ function createHtmlNormalizationState(): HtmlNormalizationState { return { lists: [], openItems: [], itemHasContent: [] }; } -const HTML_TAG_REGEX = /<\/?(?:br|p|ol|ul|li)\b(?:\s[^>]*)?\s*\/?>/gi; +const HTML_TAG_REGEX = /<\/?(?:br|p|ol|ul|li|span|text)\b(?:\s[^>]*)?\s*\/?>/gi; function htmlTagName(tag: string): string { const match = /^<\/?\s*([A-Za-z][A-Za-z0-9:-]*)/.exec(tag); @@ -96,22 +132,28 @@ function normalizeHtmlForTerminal(raw: string, state: HtmlNormalizationState = c const tag = match[0]; const index = match.index ?? 0; const textBeforeTag = normalizeHtmlEntitiesForTerminal(raw.slice(lastIndex, index)); + const name = htmlTagName(tag); + // Every tag handled here is block-level EXCEPT span and text. For block-level tags, // HTML formatting whitespace between block/list tags (e.g. the newlines and // indentation in pretty-printed `