feat(coding-agent): added HTML content extraction with Readability and GFM markdown support
- Added `extractReadableFromHtml()` utility function with dual-path content extraction using Readability library and CSS selector fallback. - Integrated Turndown library with GitHub Flavored Markdown plugin for improved HTML-to-markdown conversion supporting tables, strikethrough, and task lists. - Refactored `getPageReadable()` action to use new extraction function, consolidating content parsing logic and improving maintainability. - Added TypeScript type declarations for turndown-plugin-gfm module with custom Turndown rules for enhanced markdown formatting.
This commit is contained in:
@@ -77,10 +77,13 @@
|
||||
"lru-cache": "11.3.1",
|
||||
"markit-ai": "0.5.0",
|
||||
"puppeteer": "^24.37",
|
||||
"turndown": "7.2.4",
|
||||
"turndown-plugin-gfm": "1.0.2",
|
||||
"zod": "4.3.6",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "^1.3",
|
||||
"@types/turndown": "5.0.6",
|
||||
},
|
||||
},
|
||||
"packages/natives": {
|
||||
@@ -661,6 +664,8 @@
|
||||
|
||||
"@types/triple-beam": ["@types/triple-beam@1.3.5", "", {}, "sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw=="],
|
||||
|
||||
"@types/turndown": ["@types/turndown@5.0.6", "", {}, "sha512-ru00MoyeeouE5BX4gRL+6m/BsDfbRayOskWqUvh7CLGW+UXxHQItqALa38kKnOiZPqJrtzJUgAC2+F0rL1S4Pg=="],
|
||||
|
||||
"@types/yauzl": ["@types/yauzl@2.10.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q=="],
|
||||
|
||||
"@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260322.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260322.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-CmzQTKvesYHmz3g92G+XPDis25ocvHqa/gK8m98w+bML99KJLEWQKVlvkLrYA85JiJEK+XBIiz+6lCgUqRkWXA=="],
|
||||
|
||||
@@ -4140,7 +4140,7 @@ function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>
|
||||
.tree
|
||||
.chunks
|
||||
.iter()
|
||||
.find(|c| c.path.ends_with(".if"))
|
||||
.find(|c| Path::new(&c.path).extension().is_some_and(|ext| ext.eq_ignore_ascii_case("if")))
|
||||
.expect("if chunk should exist");
|
||||
assert!(if_chunk.leaf, "if chunk should be leaf");
|
||||
|
||||
|
||||
@@ -1,12 +1,16 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `extractReadableFromHtml` utility function to extract readable content from HTML with Readability article extraction and CSS selector fallback
|
||||
- Added support for GFM (GitHub Flavored Markdown) features including tables, strikethrough, and task lists in HTML-to-markdown conversion
|
||||
- Added `resolveDiagnosticTargets` utility function to handle glob pattern resolution with fallback to literal file paths for bracket-style paths
|
||||
|
||||
### Changed
|
||||
|
||||
- Replaced regex-based HTML-to-markdown conversion with Turndown library and GFM plugin for more accurate formatting of complex HTML structures
|
||||
- Simplified no-changes response to omit redundant response text when chunk content already matches
|
||||
- Clarified region suffix behavior on leaf and compound statement chunks — `~` and `^` now fall back to whole-chunk replacement with explicit guidance to supply complete structural content
|
||||
- Updated CRC refresh guidance to direct users to use CRCs from edit responses or run `read(path="file", sel="?")`
|
||||
|
||||
@@ -63,10 +63,13 @@
|
||||
"lru-cache": "11.3.1",
|
||||
"markit-ai": "0.5.0",
|
||||
"puppeteer": "^24.37",
|
||||
"turndown": "7.2.4",
|
||||
"turndown-plugin-gfm": "1.0.2",
|
||||
"zod": "4.3.6"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "^1.3"
|
||||
"@types/bun": "^1.3",
|
||||
"@types/turndown": "5.0.6"
|
||||
},
|
||||
"engines": {
|
||||
"bun": ">=1.3.7"
|
||||
|
||||
@@ -501,6 +501,83 @@ export interface ReadableResult {
|
||||
markdown?: string;
|
||||
}
|
||||
|
||||
type ReadableFormat = "text" | "markdown";
|
||||
|
||||
/** Trim to non-empty string or undefined. */
|
||||
function normalize(text: string | null | undefined): string | undefined {
|
||||
const trimmed = text?.trim();
|
||||
return trimmed || undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract readable content from raw HTML.
|
||||
* Tries Readability (article-isolation scoring) first, then falls back to a
|
||||
* CSS selector chain over the same pre-parsed DOM. Returns null if neither
|
||||
* path yields usable content.
|
||||
*/
|
||||
export function extractReadableFromHtml(html: string, url: string, format: ReadableFormat): ReadableResult | null {
|
||||
const { document } = parseHTML(html);
|
||||
|
||||
// --- Primary: Readability article extraction ---
|
||||
const article = new Readability(document).parse();
|
||||
if (article) {
|
||||
const result = toReadableResult(url, format, article.textContent, article.content, {
|
||||
title: article.title,
|
||||
byline: article.byline,
|
||||
excerpt: article.excerpt,
|
||||
length: article.length,
|
||||
});
|
||||
if (result) return result;
|
||||
}
|
||||
|
||||
// --- Fallback: CSS selector chain ---
|
||||
const candidates = [
|
||||
document.querySelector("[data-pagefind-body]"),
|
||||
document.querySelector("main article"),
|
||||
document.querySelector("article"),
|
||||
document.querySelector("main"),
|
||||
document.querySelector("[role='main']"),
|
||||
document.body,
|
||||
];
|
||||
for (const el of candidates) {
|
||||
if (!el) continue;
|
||||
const innerHTML = el.innerHTML?.trim();
|
||||
const textContent = el.textContent?.trim();
|
||||
if (!innerHTML || !textContent) continue;
|
||||
const result = toReadableResult(url, format, textContent, innerHTML, {
|
||||
title: document.title,
|
||||
excerpt: textContent.slice(0, 240),
|
||||
length: textContent.length,
|
||||
});
|
||||
if (result) return result;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Shared builder for both extraction paths. */
|
||||
function toReadableResult(
|
||||
url: string,
|
||||
format: ReadableFormat,
|
||||
textContent: string | null | undefined,
|
||||
htmlContent: string | null | undefined,
|
||||
meta: { title?: string | null; byline?: string | null; excerpt?: string | null; length?: number | null },
|
||||
): ReadableResult | null {
|
||||
const text = normalize(textContent);
|
||||
const markdown = format === "markdown" ? (normalize(htmlToBasicMarkdown(htmlContent ?? "")) ?? text) : undefined;
|
||||
const normalizedText = format === "text" ? text : undefined;
|
||||
if (!normalizedText && !markdown) return null;
|
||||
return {
|
||||
url,
|
||||
title: normalize(meta.title),
|
||||
byline: normalize(meta.byline),
|
||||
excerpt: normalize(meta.excerpt),
|
||||
contentLength: meta.length ?? text?.length ?? markdown?.length ?? 0,
|
||||
text: normalizedText,
|
||||
markdown,
|
||||
};
|
||||
}
|
||||
|
||||
function ensureParam<T>(value: T | undefined, name: string, action: string): T {
|
||||
if (value === undefined || value === null || value === "") {
|
||||
throw new ToolError(`Missing required parameter '${name}' for action '${action}'.`);
|
||||
@@ -1365,26 +1442,13 @@ export class BrowserTool implements AgentTool<typeof browserSchema, BrowserToolD
|
||||
const format = params.format ?? "markdown";
|
||||
const html = (await untilAborted(signal, () => page.content())) as string;
|
||||
const url = page.url();
|
||||
const { document } = parseHTML(html);
|
||||
const reader = new Readability(document);
|
||||
const article = reader.parse();
|
||||
if (!article) {
|
||||
const readable = extractReadableFromHtml(html, url, format);
|
||||
if (!readable) {
|
||||
throw new ToolError("Readable content not found");
|
||||
}
|
||||
const markdown = format === "markdown" ? htmlToBasicMarkdown(article.content ?? "") : undefined;
|
||||
const text = format === "text" ? (article.textContent ?? "") : undefined;
|
||||
const readable: ReadableResult = {
|
||||
url,
|
||||
title: article.title ?? undefined,
|
||||
byline: article.byline ?? undefined,
|
||||
excerpt: article.excerpt ?? undefined,
|
||||
contentLength: article.length ?? article.textContent?.length ?? 0,
|
||||
text,
|
||||
markdown,
|
||||
};
|
||||
details.url = url;
|
||||
details.readable = readable;
|
||||
details.result = format === "markdown" ? (markdown ?? "") : (text ?? "");
|
||||
details.result = format === "markdown" ? (readable.markdown ?? "") : (readable.text ?? "");
|
||||
return toolResult(details)
|
||||
.text(JSON.stringify(readable, null, 2))
|
||||
.done();
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
* Shared types and utilities for web-fetch handlers
|
||||
*/
|
||||
import { ptree } from "@oh-my-pi/pi-utils";
|
||||
import TurndownService from "turndown";
|
||||
import { gfm } from "turndown-plugin-gfm";
|
||||
import { ToolAbortError } from "../../tools/tool-errors";
|
||||
|
||||
export { formatNumber } from "@oh-my-pi/pi-utils";
|
||||
@@ -153,41 +155,57 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom
|
||||
return { content: "", contentType: "", finalUrl: url, ok: false };
|
||||
}
|
||||
|
||||
/** Module-level Turndown instance — matches markit-ai's configuration. */
|
||||
const turndown = new TurndownService({
|
||||
headingStyle: "atx",
|
||||
codeBlockStyle: "fenced",
|
||||
bulletListMarker: "-",
|
||||
});
|
||||
turndown.use(gfm);
|
||||
turndown.addRule("strikethrough", {
|
||||
filter: ["del", "s", "strike"],
|
||||
replacement(content) {
|
||||
return `~~${content}~~`;
|
||||
},
|
||||
});
|
||||
turndown.addRule("heading", {
|
||||
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
|
||||
replacement(content, node) {
|
||||
const level = Number(node.nodeName.charAt(1));
|
||||
const prefix = "#".repeat(level);
|
||||
const cleaned = content.replace(/\\([.])/g, "$1").trim();
|
||||
return `\n\n${prefix} ${cleaned}\n\n`;
|
||||
},
|
||||
});
|
||||
|
||||
type TurndownListParent = {
|
||||
nodeName: string;
|
||||
getAttribute(name: string): string | null;
|
||||
children: ArrayLike<unknown>;
|
||||
};
|
||||
|
||||
turndown.addRule("listItem", {
|
||||
filter: "li",
|
||||
replacement(content, node, options) {
|
||||
content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
|
||||
const parent = node.parentNode as unknown as TurndownListParent | null;
|
||||
let prefix = `${options.bulletListMarker} `;
|
||||
if (parent?.nodeName === "OL") {
|
||||
const start = parent.getAttribute("start");
|
||||
const index = Array.prototype.indexOf.call(parent.children, node);
|
||||
prefix = `${(start ? Number(start) : 1) + index}. `;
|
||||
}
|
||||
return prefix + content + (node.nextSibling ? "\n" : "");
|
||||
},
|
||||
});
|
||||
|
||||
/**
|
||||
* Convert basic HTML to markdown
|
||||
* Convert HTML to markdown using Turndown with GFM support.
|
||||
* Strips script/style tags before conversion.
|
||||
*/
|
||||
export function htmlToBasicMarkdown(html: string): string {
|
||||
const stripped = html
|
||||
.replace(/<pre[^>]*><code[^>]*>/g, "\n```\n")
|
||||
.replace(/<\/code><\/pre>/g, "\n```\n")
|
||||
.replace(/<code[^>]*>/g, "`")
|
||||
.replace(/<\/code>/g, "`")
|
||||
.replace(/<strong[^>]*>/g, "**")
|
||||
.replace(/<\/strong>/g, "**")
|
||||
.replace(/<b[^>]*>/g, "**")
|
||||
.replace(/<\/b>/g, "**")
|
||||
.replace(/<em[^>]*>/g, "*")
|
||||
.replace(/<\/em>/g, "*")
|
||||
.replace(/<i[^>]*>/g, "*")
|
||||
.replace(/<\/i>/g, "*")
|
||||
.replace(
|
||||
/<a[^>]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/g,
|
||||
(_, href, text) => `[${text.replace(/<[^>]+>/g, "").trim()}](${href})`,
|
||||
)
|
||||
.replace(/<p[^>]*>/g, "\n\n")
|
||||
.replace(/<\/p>/g, "")
|
||||
.replace(/<br\s*\/?>/g, "\n")
|
||||
.replace(/<li[^>]*>/g, "- ")
|
||||
.replace(/<\/li>/g, "\n")
|
||||
.replace(/<\/?[uo]l[^>]*>/g, "\n")
|
||||
.replace(/<h(\d)[^>]*>/g, (_, n) => `\n${"#".repeat(parseInt(n, 10))} `)
|
||||
.replace(/<\/h\d>/g, "\n")
|
||||
.replace(/<blockquote[^>]*>/g, "\n> ")
|
||||
.replace(/<\/blockquote>/g, "\n")
|
||||
.replace(/<[^>]+>/g, "")
|
||||
.replace(/\n{3,}/g, "\n\n")
|
||||
.trim();
|
||||
return decodeHtmlEntities(stripped);
|
||||
const cleaned = html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "");
|
||||
return turndown.turndown(cleaned).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { extractReadableFromHtml } from "@oh-my-pi/pi-coding-agent/tools/browser";
|
||||
|
||||
describe("browser readable extraction", () => {
|
||||
it("extracts markdown content from article-style pages", () => {
|
||||
const html = `<!doctype html>
|
||||
<html>
|
||||
<head><title>Docs</title></head>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Responses API</h1>
|
||||
<p>The Responses API stores output only when you opt in.</p>
|
||||
</article>
|
||||
</body>
|
||||
</html>`;
|
||||
|
||||
const result = extractReadableFromHtml(html, "https://example.com/docs", "markdown");
|
||||
|
||||
expect(result).not.toBeNull();
|
||||
expect(result?.title).toBe("Docs");
|
||||
expect(result?.markdown).toContain("Responses API");
|
||||
expect(result?.markdown).toContain("stores output only when you opt in");
|
||||
});
|
||||
|
||||
it("extracts docs-style main content", () => {
|
||||
const html = `<!doctype html>
|
||||
<html>
|
||||
<head><title>Reference</title></head>
|
||||
<body>
|
||||
<div class="app-shell">
|
||||
<nav>Navigation</nav>
|
||||
<main data-pagefind-body>
|
||||
<section>
|
||||
<h1>Apps SDK</h1>
|
||||
<p>Build once, run in many places.</p>
|
||||
</section>
|
||||
</main>
|
||||
</div>
|
||||
</body>
|
||||
</html>`;
|
||||
|
||||
const result = extractReadableFromHtml(html, "https://developers.openai.com/apps-sdk/reference", "text");
|
||||
|
||||
expect(result).not.toBeNull();
|
||||
expect(result?.title).toBe("Reference");
|
||||
expect(result?.text).toContain("Apps SDK");
|
||||
expect(result?.text).toContain("Build once, run in many places");
|
||||
});
|
||||
});
|
||||
Vendored
+9
@@ -12,3 +12,12 @@ declare module "*.py" {
|
||||
const content: string;
|
||||
export default content;
|
||||
}
|
||||
|
||||
// turndown-plugin-gfm has no published types
|
||||
declare module "turndown-plugin-gfm" {
|
||||
import type TurndownService from "turndown";
|
||||
export const gfm: TurndownService.Plugin;
|
||||
export const tables: TurndownService.Plugin;
|
||||
export const strikethrough: TurndownService.Plugin;
|
||||
export const taskListItems: TurndownService.Plugin;
|
||||
}
|
||||
Reference in New Issue
Block a user