feat(coding-agent): added HTML content extraction with Readability and GFM markdown support

- Added `extractReadableFromHtml()` utility function with dual-path content extraction using Readability library and CSS selector fallback.
- Integrated Turndown library with GitHub Flavored Markdown plugin for improved HTML-to-markdown conversion supporting tables, strikethrough, and task lists.
- Refactored `getPageReadable()` action to use new extraction function, consolidating content parsing logic and improving maintainability.
- Added TypeScript type declarations for turndown-plugin-gfm module with custom Turndown rules for enhanced markdown formatting.
This commit is contained in:
can1357
2026-04-10 21:18:42 +02:00
parent 7213aaf128
commit cc2801e6ef
8 changed files with 202 additions and 50 deletions
+5
View File
@@ -77,10 +77,13 @@
"lru-cache": "11.3.1",
"markit-ai": "0.5.0",
"puppeteer": "^24.37",
"turndown": "7.2.4",
"turndown-plugin-gfm": "1.0.2",
"zod": "4.3.6",
},
"devDependencies": {
"@types/bun": "^1.3",
"@types/turndown": "5.0.6",
},
},
"packages/natives": {
@@ -661,6 +664,8 @@
"@types/triple-beam": ["@types/triple-beam@1.3.5", "", {}, "sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw=="],
"@types/turndown": ["@types/turndown@5.0.6", "", {}, "sha512-ru00MoyeeouE5BX4gRL+6m/BsDfbRayOskWqUvh7CLGW+UXxHQItqALa38kKnOiZPqJrtzJUgAC2+F0rL1S4Pg=="],
"@types/yauzl": ["@types/yauzl@2.10.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q=="],
"@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260322.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260322.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-CmzQTKvesYHmz3g92G+XPDis25ocvHqa/gK8m98w+bML99KJLEWQKVlvkLrYA85JiJEK+XBIiz+6lCgUqRkWXA=="],
+1 -1
View File
@@ -4140,7 +4140,7 @@ function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>
.tree
.chunks
.iter()
.find(|c| c.path.ends_with(".if"))
.find(|c| Path::new(&c.path).extension().is_some_and(|ext| ext.eq_ignore_ascii_case("if")))
.expect("if chunk should exist");
assert!(if_chunk.leaf, "if chunk should be leaf");
+4
View File
@@ -1,12 +1,16 @@
# Changelog
## [Unreleased]
### Added
- Added `extractReadableFromHtml` utility function to extract readable content from HTML with Readability article extraction and CSS selector fallback
- Added support for GFM (GitHub Flavored Markdown) features including tables, strikethrough, and task lists in HTML-to-markdown conversion
- Added `resolveDiagnosticTargets` utility function to handle glob pattern resolution with fallback to literal file paths for bracket-style paths
### Changed
- Replaced regex-based HTML-to-markdown conversion with Turndown library and GFM plugin for more accurate formatting of complex HTML structures
- Simplified no-changes response to omit redundant response text when chunk content already matches
- Clarified region suffix behavior on leaf and compound statement chunks — `~` and `^` now fall back to whole-chunk replacement with explicit guidance to supply complete structural content
- Updated CRC refresh guidance to direct users to use CRCs from edit responses or run `read(path="file", sel="?")`
+4 -1
View File
@@ -63,10 +63,13 @@
"lru-cache": "11.3.1",
"markit-ai": "0.5.0",
"puppeteer": "^24.37",
"turndown": "7.2.4",
"turndown-plugin-gfm": "1.0.2",
"zod": "4.3.6"
},
"devDependencies": {
"@types/bun": "^1.3"
"@types/bun": "^1.3",
"@types/turndown": "5.0.6"
},
"engines": {
"bun": ">=1.3.7"
+80 -16
View File
@@ -501,6 +501,83 @@ export interface ReadableResult {
markdown?: string;
}
type ReadableFormat = "text" | "markdown";
/** Trim to non-empty string or undefined. */
function normalize(text: string | null | undefined): string | undefined {
const trimmed = text?.trim();
return trimmed || undefined;
}
/**
* Extract readable content from raw HTML.
* Tries Readability (article-isolation scoring) first, then falls back to a
* CSS selector chain over the same pre-parsed DOM. Returns null if neither
* path yields usable content.
*/
export function extractReadableFromHtml(html: string, url: string, format: ReadableFormat): ReadableResult | null {
const { document } = parseHTML(html);
// --- Primary: Readability article extraction ---
const article = new Readability(document).parse();
if (article) {
const result = toReadableResult(url, format, article.textContent, article.content, {
title: article.title,
byline: article.byline,
excerpt: article.excerpt,
length: article.length,
});
if (result) return result;
}
// --- Fallback: CSS selector chain ---
const candidates = [
document.querySelector("[data-pagefind-body]"),
document.querySelector("main article"),
document.querySelector("article"),
document.querySelector("main"),
document.querySelector("[role='main']"),
document.body,
];
for (const el of candidates) {
if (!el) continue;
const innerHTML = el.innerHTML?.trim();
const textContent = el.textContent?.trim();
if (!innerHTML || !textContent) continue;
const result = toReadableResult(url, format, textContent, innerHTML, {
title: document.title,
excerpt: textContent.slice(0, 240),
length: textContent.length,
});
if (result) return result;
}
return null;
}
/** Shared builder for both extraction paths. */
function toReadableResult(
url: string,
format: ReadableFormat,
textContent: string | null | undefined,
htmlContent: string | null | undefined,
meta: { title?: string | null; byline?: string | null; excerpt?: string | null; length?: number | null },
): ReadableResult | null {
const text = normalize(textContent);
const markdown = format === "markdown" ? (normalize(htmlToBasicMarkdown(htmlContent ?? "")) ?? text) : undefined;
const normalizedText = format === "text" ? text : undefined;
if (!normalizedText && !markdown) return null;
return {
url,
title: normalize(meta.title),
byline: normalize(meta.byline),
excerpt: normalize(meta.excerpt),
contentLength: meta.length ?? text?.length ?? markdown?.length ?? 0,
text: normalizedText,
markdown,
};
}
function ensureParam<T>(value: T | undefined, name: string, action: string): T {
if (value === undefined || value === null || value === "") {
throw new ToolError(`Missing required parameter '${name}' for action '${action}'.`);
@@ -1365,26 +1442,13 @@ export class BrowserTool implements AgentTool<typeof browserSchema, BrowserToolD
const format = params.format ?? "markdown";
const html = (await untilAborted(signal, () => page.content())) as string;
const url = page.url();
const { document } = parseHTML(html);
const reader = new Readability(document);
const article = reader.parse();
if (!article) {
const readable = extractReadableFromHtml(html, url, format);
if (!readable) {
throw new ToolError("Readable content not found");
}
const markdown = format === "markdown" ? htmlToBasicMarkdown(article.content ?? "") : undefined;
const text = format === "text" ? (article.textContent ?? "") : undefined;
const readable: ReadableResult = {
url,
title: article.title ?? undefined,
byline: article.byline ?? undefined,
excerpt: article.excerpt ?? undefined,
contentLength: article.length ?? article.textContent?.length ?? 0,
text,
markdown,
};
details.url = url;
details.readable = readable;
details.result = format === "markdown" ? (markdown ?? "") : (text ?? "");
details.result = format === "markdown" ? (readable.markdown ?? "") : (readable.text ?? "");
return toolResult(details)
.text(JSON.stringify(readable, null, 2))
.done();
+50 -32
View File
@@ -2,6 +2,8 @@
* Shared types and utilities for web-fetch handlers
*/
import { ptree } from "@oh-my-pi/pi-utils";
import TurndownService from "turndown";
import { gfm } from "turndown-plugin-gfm";
import { ToolAbortError } from "../../tools/tool-errors";
export { formatNumber } from "@oh-my-pi/pi-utils";
@@ -153,41 +155,57 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom
return { content: "", contentType: "", finalUrl: url, ok: false };
}
/** Module-level Turndown instance — matches markit-ai's configuration. */
const turndown = new TurndownService({
headingStyle: "atx",
codeBlockStyle: "fenced",
bulletListMarker: "-",
});
turndown.use(gfm);
turndown.addRule("strikethrough", {
filter: ["del", "s", "strike"],
replacement(content) {
return `~~${content}~~`;
},
});
turndown.addRule("heading", {
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
replacement(content, node) {
const level = Number(node.nodeName.charAt(1));
const prefix = "#".repeat(level);
const cleaned = content.replace(/\\([.])/g, "$1").trim();
return `\n\n${prefix} ${cleaned}\n\n`;
},
});
type TurndownListParent = {
nodeName: string;
getAttribute(name: string): string | null;
children: ArrayLike<unknown>;
};
turndown.addRule("listItem", {
filter: "li",
replacement(content, node, options) {
content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
const parent = node.parentNode as unknown as TurndownListParent | null;
let prefix = `${options.bulletListMarker} `;
if (parent?.nodeName === "OL") {
const start = parent.getAttribute("start");
const index = Array.prototype.indexOf.call(parent.children, node);
prefix = `${(start ? Number(start) : 1) + index}. `;
}
return prefix + content + (node.nextSibling ? "\n" : "");
},
});
/**
* Convert basic HTML to markdown
* Convert HTML to markdown using Turndown with GFM support.
* Strips script/style tags before conversion.
*/
export function htmlToBasicMarkdown(html: string): string {
const stripped = html
.replace(/<pre[^>]*><code[^>]*>/g, "\n```\n")
.replace(/<\/code><\/pre>/g, "\n```\n")
.replace(/<code[^>]*>/g, "`")
.replace(/<\/code>/g, "`")
.replace(/<strong[^>]*>/g, "**")
.replace(/<\/strong>/g, "**")
.replace(/<b[^>]*>/g, "**")
.replace(/<\/b>/g, "**")
.replace(/<em[^>]*>/g, "*")
.replace(/<\/em>/g, "*")
.replace(/<i[^>]*>/g, "*")
.replace(/<\/i>/g, "*")
.replace(
/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/g,
(_, href, text) => `[${text.replace(/<[^>]+>/g, "").trim()}](${href})`,
)
.replace(/<p[^>]*>/g, "\n\n")
.replace(/<\/p>/g, "")
.replace(/<br\s*\/?>/g, "\n")
.replace(/<li[^>]*>/g, "- ")
.replace(/<\/li>/g, "\n")
.replace(/<\/?[uo]l[^>]*>/g, "\n")
.replace(/<h(\d)[^>]*>/g, (_, n) => `\n${"#".repeat(parseInt(n, 10))} `)
.replace(/<\/h\d>/g, "\n")
.replace(/<blockquote[^>]*>/g, "\n> ")
.replace(/<\/blockquote>/g, "\n")
.replace(/<[^>]+>/g, "")
.replace(/\n{3,}/g, "\n\n")
.trim();
return decodeHtmlEntities(stripped);
const cleaned = html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "");
return turndown.turndown(cleaned).trim();
}
/**
@@ -0,0 +1,49 @@
import { describe, expect, it } from "bun:test";
import { extractReadableFromHtml } from "@oh-my-pi/pi-coding-agent/tools/browser";
describe("browser readable extraction", () => {
it("extracts markdown content from article-style pages", () => {
const html = `<!doctype html>
<html>
<head><title>Docs</title></head>
<body>
<article>
<h1>Responses API</h1>
<p>The Responses API stores output only when you opt in.</p>
</article>
</body>
</html>`;
const result = extractReadableFromHtml(html, "https://example.com/docs", "markdown");
expect(result).not.toBeNull();
expect(result?.title).toBe("Docs");
expect(result?.markdown).toContain("Responses API");
expect(result?.markdown).toContain("stores output only when you opt in");
});
it("extracts docs-style main content", () => {
const html = `<!doctype html>
<html>
<head><title>Reference</title></head>
<body>
<div class="app-shell">
<nav>Navigation</nav>
<main data-pagefind-body>
<section>
<h1>Apps SDK</h1>
<p>Build once, run in many places.</p>
</section>
</main>
</div>
</body>
</html>`;
const result = extractReadableFromHtml(html, "https://developers.openai.com/apps-sdk/reference", "text");
expect(result).not.toBeNull();
expect(result?.title).toBe("Reference");
expect(result?.text).toContain("Apps SDK");
expect(result?.text).toContain("Build once, run in many places");
});
});
+9
View File
@@ -12,3 +12,12 @@ declare module "*.py" {
const content: string;
export default content;
}
// turndown-plugin-gfm has no published types
declare module "turndown-plugin-gfm" {
import type TurndownService from "turndown";
export const gfm: TurndownService.Plugin;
export const tables: TurndownService.Plugin;
export const strikethrough: TurndownService.Plugin;
export const taskListItems: TurndownService.Plugin;
}