fix(utils): stop indented code interrupting paragraphs in marked lexer

This commit is contained in:
chan1103
2026-08-14 22:10:13 +09:00
parent ffd53ff92a
commit dc0e2ebeed
5 changed files with 100 additions and 2 deletions
+21
View File
@@ -302,6 +302,27 @@ Average Latency: 1,240 ms
expect(plainLines.filter(line => line.includes("+"))).toHaveLength(0);
});
it("keeps a four-space-indented tree child inside its paragraph", () => {
const markdown = new Markdown(
`Two verified cases:
├── **case 1** — first branch
│ └── each family has its own counter
└── **case 3: unnumbered limits**
└── limits that carry **only a label** are dropped from the list`,
0,
0,
defaultMarkdownTheme,
);
const plainLines = markdown.render(80).map(line => stripVTControlCharacters(line).trimEnd());
// The attached indented line is a lazy paragraph continuation, never an
// indented code block: no fence border rows, no literal bold markers.
expect(plainLines.filter(line => line.includes("```"))).toHaveLength(0);
expect(plainLines.some(line => line.includes("only a label"))).toBe(true);
expect(plainLines.some(line => line.includes("**"))).toBe(false);
});
it("keeps an unfinished code block with a rule and heading literal when no table follows", () => {
const markdown = new Markdown(
`The process printed this
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed the vendored Markdown lexer letting an indented code block interrupt a paragraph: a 4-space-indented line directly attached to paragraph text (e.g. a box-drawing tree child under a `└──` branch) now stays a lazy continuation of that paragraph, matching marked/CommonMark, instead of rendering the rest of the message section as a raw code block ([#8557](https://github.com/can1357/oh-my-pi/pull/8557) by [@chan1103](https://github.com/chan1103)).
## [17.3.2] - 2026-08-13
### Fixed
+14 -2
View File
@@ -1071,8 +1071,20 @@ function blockTokens(src: string, lexer: Lexer, output: Token[]): Token[] {
}
let raw = line;
i++;
while (i < lines.length && !isBlockStart(lines, i)) {
raw += lines[i++]!;
// An indented code block cannot interrupt a paragraph (CommonMark lazy
// continuation): a 4-space-indented line directly attached to nonblank
// paragraph text stays inside the paragraph, bypassing every block-start
// probe — matching marked's paragraph rule. After a whitespace-padded
// blank line the indent is no longer attached, so it still opens an
// indented code block.
let prevBlankish = false;
while (i < lines.length) {
const next = lines[i]!;
const lazyIndent = !prevBlankish && /^ {4}\S/.test(next);
if (!lazyIndent && isBlockStart(lines, i)) break;
prevBlankish = next.trim() === "";
raw += next;
i++;
}
const text = stripFinalNewline(raw);
const tokens: Token[] = [];
+30
View File
@@ -954,5 +954,35 @@
}
],
"html": "<div>\nraw *html*\n</div><p>inline <span>x</span> end</p>\n"
},
{
"name": "lazyContinuation",
"source": "tree root\n└── last branch\n └── child under last branch\n\n real indented code\n",
"tokens": [
{
"type": "paragraph",
"raw": "tree root\n└── last branch\n └── child under last branch",
"text": "tree root\n└── last branch\n └── child under last branch",
"tokens": [
{
"type": "text",
"raw": "tree root\n└── last branch\n └── child under last branch",
"text": "tree root\n└── last branch\n └── child under last branch",
"escaped": false
}
]
},
{
"type": "space",
"raw": "\n\n"
},
{
"type": "code",
"raw": " real indented code\n",
"codeBlockStyle": "indented",
"text": "real indented code\n"
}
],
"html": "<p>tree root\n└── last branch\n └── child under last branch</p>\n<pre><code>real indented code\n</code></pre>\n"
}
]
+31
View File
@@ -81,6 +81,37 @@ describe("marked compatibility", () => {
});
}
// Lazy-continuation boundary shapes (behavior cross-checked against marked
// v18): an indented code block cannot interrupt a paragraph, so an indented
// line directly attached to paragraph text stays in the paragraph even when
// a setext/hr lookahead matches downstream — while a whitespace-padded
// blank line detaches it, so the next indented run still opens indented
// code. Documented divergences from marked kept as-is: marked dedents the
// attached line inside `text` via its code-merge path, and it splits a
// padded blank into a `space` token where this lexer keeps the padded
// blank inside the paragraph raw.
const lazyBoundaryShapes: Array<[string, Array<[string, string]>]> = [
[
"lead\n \n code\n",
[
["paragraph", "lead\n \n"],
["code", " code\n"],
],
],
[
"lead\n attached\n---\n",
[
["paragraph", "lead\n attached\n"],
["hr", "---\n"],
],
],
];
for (const [source, shape] of lazyBoundaryShapes) {
test(`keeps the lazy-continuation boundary shape for ${JSON.stringify(source)}`, () => {
expect([...Lexer.lex(source)].map(token => [token.type, token.raw])).toEqual(shape);
});
}
test("runs block and inline tokenizer/renderer extensions", () => {
const latexBlock: TokenizerAndRendererExtension = {
name: "latexBlock",