diff --git a/bun.lock b/bun.lock index a58c07137..a8acec978 100644 --- a/bun.lock +++ b/bun.lock @@ -77,10 +77,13 @@ "lru-cache": "11.3.1", "markit-ai": "0.5.0", "puppeteer": "^24.37", + "turndown": "7.2.4", + "turndown-plugin-gfm": "1.0.2", "zod": "4.3.6", }, "devDependencies": { "@types/bun": "^1.3", + "@types/turndown": "5.0.6", }, }, "packages/natives": { @@ -661,6 +664,8 @@ "@types/triple-beam": ["@types/triple-beam@1.3.5", "", {}, "sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw=="], + "@types/turndown": ["@types/turndown@5.0.6", "", {}, "sha512-ru00MoyeeouE5BX4gRL+6m/BsDfbRayOskWqUvh7CLGW+UXxHQItqALa38kKnOiZPqJrtzJUgAC2+F0rL1S4Pg=="], + "@types/yauzl": ["@types/yauzl@2.10.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q=="], "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260322.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260322.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-CmzQTKvesYHmz3g92G+XPDis25ocvHqa/gK8m98w+bML99KJLEWQKVlvkLrYA85JiJEK+XBIiz+6lCgUqRkWXA=="], diff --git a/crates/pi-natives/src/chunk/edit.rs b/crates/pi-natives/src/chunk/edit.rs index 82dd73a18..9d69871ed 100644 --- a/crates/pi-natives/src/chunk/edit.rs +++ b/crates/pi-natives/src/chunk/edit.rs @@ -4140,7 +4140,7 @@ function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>> .tree .chunks .iter() - .find(|c| c.path.ends_with(".if")) + .find(|c| Path::new(&c.path).extension().is_some_and(|ext| ext.eq_ignore_ascii_case("if"))) .expect("if chunk should exist"); assert!(if_chunk.leaf, "if chunk should be leaf"); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6424c8cc9..db56136de 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,12 +1,16 @@ # Changelog ## [Unreleased] + ### Added +- Added `extractReadableFromHtml` utility function to extract readable content from HTML with Readability article extraction and CSS selector fallback +- Added support for GFM (GitHub Flavored Markdown) features including tables, strikethrough, and task lists in HTML-to-markdown conversion - Added `resolveDiagnosticTargets` utility function to handle glob pattern resolution with fallback to literal file paths for bracket-style paths ### Changed +- Replaced regex-based HTML-to-markdown conversion with Turndown library and GFM plugin for more accurate formatting of complex HTML structures - Simplified no-changes response to omit redundant response text when chunk content already matches - Clarified region suffix behavior on leaf and compound statement chunks — `~` and `^` now fall back to whole-chunk replacement with explicit guidance to supply complete structural content - Updated CRC refresh guidance to direct users to use CRCs from edit responses or run `read(path="file", sel="?")` diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 6ef7861ab..f13708e8e 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -63,10 +63,13 @@ "lru-cache": "11.3.1", "markit-ai": "0.5.0", "puppeteer": "^24.37", + "turndown": "7.2.4", + "turndown-plugin-gfm": "1.0.2", "zod": "4.3.6" }, "devDependencies": { - "@types/bun": "^1.3" + "@types/bun": "^1.3", + "@types/turndown": "5.0.6" }, "engines": { "bun": ">=1.3.7" diff --git a/packages/coding-agent/src/tools/browser.ts b/packages/coding-agent/src/tools/browser.ts index 6469d965e..58da91bad 100644 --- a/packages/coding-agent/src/tools/browser.ts +++ b/packages/coding-agent/src/tools/browser.ts @@ -501,6 +501,83 @@ export interface ReadableResult { markdown?: string; } +type ReadableFormat = "text" | "markdown"; + +/** Trim to non-empty string or undefined. */ +function normalize(text: string | null | undefined): string | undefined { + const trimmed = text?.trim(); + return trimmed || undefined; +} + +/** + * Extract readable content from raw HTML. + * Tries Readability (article-isolation scoring) first, then falls back to a + * CSS selector chain over the same pre-parsed DOM. Returns null if neither + * path yields usable content. + */ +export function extractReadableFromHtml(html: string, url: string, format: ReadableFormat): ReadableResult | null { + const { document } = parseHTML(html); + + // --- Primary: Readability article extraction --- + const article = new Readability(document).parse(); + if (article) { + const result = toReadableResult(url, format, article.textContent, article.content, { + title: article.title, + byline: article.byline, + excerpt: article.excerpt, + length: article.length, + }); + if (result) return result; + } + + // --- Fallback: CSS selector chain --- + const candidates = [ + document.querySelector("[data-pagefind-body]"), + document.querySelector("main article"), + document.querySelector("article"), + document.querySelector("main"), + document.querySelector("[role='main']"), + document.body, + ]; + for (const el of candidates) { + if (!el) continue; + const innerHTML = el.innerHTML?.trim(); + const textContent = el.textContent?.trim(); + if (!innerHTML || !textContent) continue; + const result = toReadableResult(url, format, textContent, innerHTML, { + title: document.title, + excerpt: textContent.slice(0, 240), + length: textContent.length, + }); + if (result) return result; + } + + return null; +} + +/** Shared builder for both extraction paths. */ +function toReadableResult( + url: string, + format: ReadableFormat, + textContent: string | null | undefined, + htmlContent: string | null | undefined, + meta: { title?: string | null; byline?: string | null; excerpt?: string | null; length?: number | null }, +): ReadableResult | null { + const text = normalize(textContent); + const markdown = format === "markdown" ? (normalize(htmlToBasicMarkdown(htmlContent ?? "")) ?? text) : undefined; + const normalizedText = format === "text" ? text : undefined; + if (!normalizedText && !markdown) return null; + return { + url, + title: normalize(meta.title), + byline: normalize(meta.byline), + excerpt: normalize(meta.excerpt), + contentLength: meta.length ?? text?.length ?? markdown?.length ?? 0, + text: normalizedText, + markdown, + }; +} + function ensureParam(value: T | undefined, name: string, action: string): T { if (value === undefined || value === null || value === "") { throw new ToolError(`Missing required parameter '${name}' for action '${action}'.`); @@ -1365,26 +1442,13 @@ export class BrowserTool implements AgentTool page.content())) as string; const url = page.url(); - const { document } = parseHTML(html); - const reader = new Readability(document); - const article = reader.parse(); - if (!article) { + const readable = extractReadableFromHtml(html, url, format); + if (!readable) { throw new ToolError("Readable content not found"); } - const markdown = format === "markdown" ? htmlToBasicMarkdown(article.content ?? "") : undefined; - const text = format === "text" ? (article.textContent ?? "") : undefined; - const readable: ReadableResult = { - url, - title: article.title ?? undefined, - byline: article.byline ?? undefined, - excerpt: article.excerpt ?? undefined, - contentLength: article.length ?? article.textContent?.length ?? 0, - text, - markdown, - }; details.url = url; details.readable = readable; - details.result = format === "markdown" ? (markdown ?? "") : (text ?? ""); + details.result = format === "markdown" ? (readable.markdown ?? "") : (readable.text ?? ""); return toolResult(details) .text(JSON.stringify(readable, null, 2)) .done(); diff --git a/packages/coding-agent/src/web/scrapers/types.ts b/packages/coding-agent/src/web/scrapers/types.ts index 9181c9880..0f48d87e0 100644 --- a/packages/coding-agent/src/web/scrapers/types.ts +++ b/packages/coding-agent/src/web/scrapers/types.ts @@ -2,6 +2,8 @@ * Shared types and utilities for web-fetch handlers */ import { ptree } from "@oh-my-pi/pi-utils"; +import TurndownService from "turndown"; +import { gfm } from "turndown-plugin-gfm"; import { ToolAbortError } from "../../tools/tool-errors"; export { formatNumber } from "@oh-my-pi/pi-utils"; @@ -153,41 +155,57 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom return { content: "", contentType: "", finalUrl: url, ok: false }; } +/** Module-level Turndown instance — matches markit-ai's configuration. */ +const turndown = new TurndownService({ + headingStyle: "atx", + codeBlockStyle: "fenced", + bulletListMarker: "-", +}); +turndown.use(gfm); +turndown.addRule("strikethrough", { + filter: ["del", "s", "strike"], + replacement(content) { + return `~~${content}~~`; + }, +}); +turndown.addRule("heading", { + filter: ["h1", "h2", "h3", "h4", "h5", "h6"], + replacement(content, node) { + const level = Number(node.nodeName.charAt(1)); + const prefix = "#".repeat(level); + const cleaned = content.replace(/\\([.])/g, "$1").trim(); + return `\n\n${prefix} ${cleaned}\n\n`; + }, +}); + +type TurndownListParent = { + nodeName: string; + getAttribute(name: string): string | null; + children: ArrayLike; +}; + +turndown.addRule("listItem", { + filter: "li", + replacement(content, node, options) { + content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n "); + const parent = node.parentNode as unknown as TurndownListParent | null; + let prefix = `${options.bulletListMarker} `; + if (parent?.nodeName === "OL") { + const start = parent.getAttribute("start"); + const index = Array.prototype.indexOf.call(parent.children, node); + prefix = `${(start ? Number(start) : 1) + index}. `; + } + return prefix + content + (node.nextSibling ? "\n" : ""); + }, +}); + /** - * Convert basic HTML to markdown + * Convert HTML to markdown using Turndown with GFM support. + * Strips script/style tags before conversion. */ export function htmlToBasicMarkdown(html: string): string { - const stripped = html - .replace(/]*>]*>/g, "\n```\n") - .replace(/<\/code><\/pre>/g, "\n```\n") - .replace(/]*>/g, "`") - .replace(/<\/code>/g, "`") - .replace(/]*>/g, "**") - .replace(/<\/strong>/g, "**") - .replace(/]*>/g, "**") - .replace(/<\/b>/g, "**") - .replace(/]*>/g, "*") - .replace(/<\/em>/g, "*") - .replace(/]*>/g, "*") - .replace(/<\/i>/g, "*") - .replace( - /]*href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/g, - (_, href, text) => `[${text.replace(/<[^>]+>/g, "").trim()}](${href})`, - ) - .replace(/]*>/g, "\n\n") - .replace(/<\/p>/g, "") - .replace(//g, "\n") - .replace(/]*>/g, "- ") - .replace(/<\/li>/g, "\n") - .replace(/<\/?[uo]l[^>]*>/g, "\n") - .replace(/]*>/g, (_, n) => `\n${"#".repeat(parseInt(n, 10))} `) - .replace(/<\/h\d>/g, "\n") - .replace(/]*>/g, "\n> ") - .replace(/<\/blockquote>/g, "\n") - .replace(/<[^>]+>/g, "") - .replace(/\n{3,}/g, "\n\n") - .trim(); - return decodeHtmlEntities(stripped); + const cleaned = html.replace(//gi, "").replace(//gi, ""); + return turndown.turndown(cleaned).trim(); } /** diff --git a/packages/coding-agent/test/tools/browser-readable.test.ts b/packages/coding-agent/test/tools/browser-readable.test.ts new file mode 100644 index 000000000..7f8d2c5d9 --- /dev/null +++ b/packages/coding-agent/test/tools/browser-readable.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from "bun:test"; +import { extractReadableFromHtml } from "@oh-my-pi/pi-coding-agent/tools/browser"; + +describe("browser readable extraction", () => { + it("extracts markdown content from article-style pages", () => { + const html = ` + + Docs + +
+

Responses API

+

The Responses API stores output only when you opt in.

+
+ + `; + + const result = extractReadableFromHtml(html, "https://example.com/docs", "markdown"); + + expect(result).not.toBeNull(); + expect(result?.title).toBe("Docs"); + expect(result?.markdown).toContain("Responses API"); + expect(result?.markdown).toContain("stores output only when you opt in"); + }); + + it("extracts docs-style main content", () => { + const html = ` + + Reference + +
+ +
+
+

Apps SDK

+

Build once, run in many places.

+
+
+
+ + `; + + const result = extractReadableFromHtml(html, "https://developers.openai.com/apps-sdk/reference", "text"); + + expect(result).not.toBeNull(); + expect(result?.title).toBe("Reference"); + expect(result?.text).toContain("Apps SDK"); + expect(result?.text).toContain("Build once, run in many places"); + }); +}); diff --git a/types/assets/index.d.ts b/types/assets/index.d.ts index 5ed830d6b..10841c39f 100644 --- a/types/assets/index.d.ts +++ b/types/assets/index.d.ts @@ -12,3 +12,12 @@ declare module "*.py" { const content: string; export default content; } + +// turndown-plugin-gfm has no published types +declare module "turndown-plugin-gfm" { + import type TurndownService from "turndown"; + export const gfm: TurndownService.Plugin; + export const tables: TurndownService.Plugin; + export const strikethrough: TurndownService.Plugin; + export const taskListItems: TurndownService.Plugin; +} \ No newline at end of file