diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e4f4f17f8..2c88a5b7d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +### Changed + +- Moved llms.txt endpoint discovery to fallback strategy when rendered page content is low quality, prioritizing page-specific content over site-wide files +- Enhanced llms.txt endpoint detection to scope candidates to the requested URL path, searching section-specific files before site-wide ones ## [13.9.6] - 2026-03-08 diff --git a/packages/coding-agent/src/tools/fetch.ts b/packages/coding-agent/src/tools/fetch.ts index 1dabe3206..6b8a4e46d 100644 --- a/packages/coding-agent/src/tools/fetch.ts +++ b/packages/coding-agent/src/tools/fetch.ts @@ -97,14 +97,28 @@ function hasCommand(cmd: string): boolean { } /** - * Extract origin from URL + * Build llms.txt candidates scoped to the requested URL */ -function getOrigin(url: string): string { +function buildLlmEndpointCandidates(url: string): string[] { try { const parsed = new URL(url); - return `${parsed.protocol}//${parsed.host}`; + if (parsed.pathname === "/") { + return [`${parsed.origin}/.well-known/llms.txt`, `${parsed.origin}/llms.txt`, `${parsed.origin}/llms.md`]; + } + + const trimmedPath = parsed.pathname.replace(/\/+$/, ""); + const segments = trimmedPath.split("/").filter(Boolean); + const scopeDepth = parsed.pathname.endsWith("/") ? segments.length : Math.max(segments.length - 1, 1); + const endpoints: string[] = []; + + for (let depth = scopeDepth; depth >= 1; depth--) { + const scope = `/${segments.slice(0, depth).join("/")}/`; + endpoints.push(`${parsed.origin}${scope}llms.txt`, `${parsed.origin}${scope}llms.md`); + } + + return endpoints; } catch { - return ""; + return []; } } @@ -227,10 +241,14 @@ async function tryMdSuffix(url: string, timeout: number, signal?: AbortSignal): /** * Try to fetch LLM-friendly endpoints */ -async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSignal): Promise { - const endpoints = [`${origin}/.well-known/llms.txt`, `${origin}/llms.txt`, `${origin}/llms.md`]; +async function tryLlmEndpoints( + url: string, + timeout: number, + signal?: AbortSignal, +): Promise<{ content: string; endpoint: string } | null> { + const endpoints = buildLlmEndpointCandidates(url); - if (signal?.aborted) { + if (signal?.aborted || endpoints.length === 0) { return null; } @@ -240,7 +258,7 @@ async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSi } const result = await loadPage(endpoint, { timeout: Math.min(timeout, 5), signal }); if (result.ok && result.content.trim().length > 100 && !looksLikeHtml(result.content)) { - return result.content; + return { content: result.content, endpoint }; } } return null; @@ -634,7 +652,6 @@ async function renderUrl( // Step 0: Normalize URL (ensure scheme for special handlers) url = normalizeUrl(url); - const origin = getOrigin(url); // Step 1: Try special handlers for known sites (unless raw mode) if (!raw) { @@ -912,24 +929,7 @@ async function renderUrl( }; } - // 5C: LLM-friendly endpoints - const llmContent = await tryLlmEndpoints(origin, timeout, signal); - if (llmContent) { - notes.push("Found llms.txt"); - const output = finalizeOutput(llmContent); - return { - url, - finalUrl, - contentType: "text/plain", - method: "llms.txt", - content: output.content, - fetchedAt, - truncated: output.truncated, - notes, - }; - } - - // 5D: Content negotiation + // 5C: Content negotiation const negotiated = await tryContentNegotiation(url, timeout, signal); if (negotiated) { notes.push(`Content negotiation returned ${negotiated.type}`); @@ -946,7 +946,7 @@ async function renderUrl( }; } - // 5E: Check for feed alternates + // 5D: Check for feed alternates const feedAlternates = alternates.filter(alt => !alt.endsWith(".md") && !alt.includes("markdown")); for (const altUrl of feedAlternates.slice(0, 2)) { const resolved = altUrl.startsWith("http") ? altUrl : new URL(altUrl, finalUrl).href; @@ -972,7 +972,7 @@ async function renderUrl( throw new ToolAbortError(); } - // Step 6: Render HTML with lynx or html2text + // 5E: Render HTML with lynx or html2text const htmlResult = await renderHtmlToText(finalUrl, rawContent, timeout, useKagiSummarizer, signal); if (!htmlResult.ok) { notes.push("html rendering failed (lynx/html2text unavailable)"); @@ -989,7 +989,7 @@ async function renderUrl( }; } - // Step 7: If lynx output is low quality, try extracting document links + // Step 6: If rendered output is low quality, try more targeted fallbacks if (isLowQualityOutput(htmlResult.content)) { const docLinks = extractDocumentLinks(rawContent, finalUrl); if (docLinks.length > 0) { @@ -1019,6 +1019,23 @@ async function renderUrl( notes.push(`Binary fetch failed: ${binary.error}`); } } + + const llmResult = await tryLlmEndpoints(finalUrl, timeout, signal); + if (llmResult) { + notes.push(`Used llms.txt fallback: ${llmResult.endpoint}`); + const output = finalizeOutput(llmResult.content); + return { + url, + finalUrl, + contentType: "text/plain", + method: "llms.txt", + content: output.content, + fetchedAt, + truncated: output.truncated, + notes, + }; + } + notes.push("Page appears to require JavaScript or is mostly navigation"); } diff --git a/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts b/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts index db1af7dcd..baae14756 100644 --- a/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts +++ b/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts @@ -11,7 +11,8 @@ import * as kagi from "@oh-my-pi/pi-coding-agent/web/kagi"; import type { LoadPageResult } from "@oh-my-pi/pi-coding-agent/web/scrapers/types"; import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types"; import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils"; -import { Snowflake } from "@oh-my-pi/pi-utils"; +import * as natives from "@oh-my-pi/pi-natives"; +import { ptree, Snowflake } from "@oh-my-pi/pi-utils"; describe("fetch tool Kagi summarization toggle", () => { let testDir: string; @@ -375,4 +376,154 @@ describe("fetch tool Kagi summarization toggle", () => { expect(textBlock?.type).toBe("text"); expect(textBlock?.text).toContain("gateway error"); }); + it("prefers rendered page content over site-wide llms.txt for deep pages", async () => { + const session = createSession({ "fetch.useKagiSummarizer": false }); + const tool = new FetchTool(session); + const pageUrl = "https://bun.com/reference/bun/UnixSocketOptions"; + const pageHtml = "

UnixSocketOptions

Page-specific docs.

"; + const renderedMarkdown = `# UnixSocketOptions\n\n${"Page-specific API docs. ".repeat(8)}`; + const originalWhich = Bun.which; + + Bun.which = (() => null) as typeof Bun.which; + try { + const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => { + if (requestedUrl === pageUrl) { + return { + ok: true, + status: 200, + contentType: "text/html", + finalUrl: pageUrl, + content: pageHtml, + }; + } + + if (requestedUrl === `${pageUrl}.md`) { + return { + ok: false, + status: 404, + contentType: "text/plain", + finalUrl: requestedUrl, + content: "", + }; + } + + if (requestedUrl === "https://bun.com/llms.txt") { + return { + ok: true, + status: 200, + contentType: "text/plain", + finalUrl: requestedUrl, + content: `# Bun\n\n${"Site-wide overview. ".repeat(12)}`, + }; + } + + return { + ok: false, + status: 404, + contentType: "text/plain", + finalUrl: requestedUrl, + content: "", + }; + }); + vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" })); + vi.spyOn(toolsManager, "ensureTool").mockResolvedValue(undefined); + vi.spyOn(natives, "htmlToMarkdown").mockResolvedValue(renderedMarkdown); + + const result = await tool.execute("fetch-deep-page", { url: pageUrl }); + const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl); + const textBlock = result.content.find(content => content.type === "text"); + + expect(result.details?.method).toBe("native"); + expect(textBlock?.type).toBe("text"); + expect(textBlock?.text).toContain("UnixSocketOptions"); + expect(requestedUrls).not.toContain("https://bun.com/.well-known/llms.txt"); + expect(requestedUrls).not.toContain("https://bun.com/llms.txt"); + expect(requestedUrls).not.toContain("https://bun.com/llms.md"); + } finally { + Bun.which = originalWhich; + } + }); + + it("uses section-scoped llms.txt fallback without requesting the site-wide file", async () => { + const session = createSession({ "fetch.useKagiSummarizer": false }); + const tool = new FetchTool(session); + const pageUrl = "https://example.com/docs/reference/widget"; + const pageHtml = "

Widget

"; + const lowQualityRender = `${"Please enable JavaScript to view this page.\n".repeat(6)}${"navigation\n".repeat(4)}`; + const originalWhich = Bun.which; + const execSpy = vi.spyOn(ptree, "exec").mockResolvedValue({ ok: true, stdout: lowQualityRender } as never); + + Bun.which = (() => null) as typeof Bun.which; + try { + const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => { + if (requestedUrl === pageUrl) { + return { + ok: true, + status: 200, + contentType: "text/html", + finalUrl: pageUrl, + content: pageHtml, + }; + } + + if ([`${pageUrl}.md`, "https://example.com/docs/reference/llms.txt", "https://example.com/docs/reference/llms.md", "https://example.com/docs/llms.md"].includes(requestedUrl)) { + return { + ok: false, + status: 404, + contentType: "text/plain", + finalUrl: requestedUrl, + content: "", + }; + } + + if (requestedUrl === "https://example.com/docs/llms.txt") { + return { + ok: true, + status: 200, + contentType: "text/plain", + finalUrl: requestedUrl, + content: `# Example Docs\n\n${"Section-scoped fallback. ".repeat(10)}`, + }; + } + + if (requestedUrl === "https://example.com/llms.txt") { + return { + ok: true, + status: 200, + contentType: "text/plain", + finalUrl: requestedUrl, + content: `# Example\n\n${"Site-wide fallback. ".repeat(10)}`, + }; + } + + return { + ok: false, + status: 404, + contentType: "text/plain", + finalUrl: requestedUrl, + content: "", + }; + }); + vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" })); + vi.spyOn(toolsManager, "ensureTool").mockResolvedValue("/usr/bin/trafilatura"); + + const result = await tool.execute("fetch-section-llms", { url: pageUrl }); + const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl); + const textBlock = result.content.find(content => content.type === "text"); + + expect(result.details?.method).toBe("llms.txt"); + expect(result.details?.notes).toContain("Used llms.txt fallback: https://example.com/docs/llms.txt"); + expect(textBlock?.type).toBe("text"); + expect(textBlock?.text).toContain("Section-scoped fallback"); + expect(requestedUrls).toContain("https://example.com/docs/llms.txt"); + expect(requestedUrls).not.toContain("https://example.com/.well-known/llms.txt"); + expect(requestedUrls).not.toContain("https://example.com/llms.txt"); + expect(requestedUrls).not.toContain("https://example.com/llms.md"); + } finally { + Bun.which = originalWhich; + execSpy.mockRestore(); + } + }); + + });