feat(coding-agent/tools): implemented path-scoped llms.txt discovery with content-quality fallback

- Moved llms.txt endpoint discovery to fallback strategy when rendered page content is low quality, prioritizing page-specific content over site-wide files.
- Enhanced llms.txt endpoint detection to scope candidates to the requested URL path, searching section-specific files before site-wide ones.
- Replaced getOrigin() with buildLlmEndpointCandidates() to generate path-scoped endpoint candidates with depth-based fallback strategy.
- Updated tryLlmEndpoints() to accept full URL and return endpoint metadata alongside content for better fallback tracking.
- Added 2 integration tests validating section-scoped llms.txt discovery and preference for rendered content over site-wide files.
This commit is contained in:
can1357
2026-03-08 04:36:05 +01:00
parent c3d83c2003
commit 73355695f2
3 changed files with 203 additions and 31 deletions
+4
View File
@@ -1,6 +1,10 @@
# Changelog
## [Unreleased]
### Changed
- Moved llms.txt endpoint discovery to fallback strategy when rendered page content is low quality, prioritizing page-specific content over site-wide files
- Enhanced llms.txt endpoint detection to scope candidates to the requested URL path, searching section-specific files before site-wide ones
## [13.9.6] - 2026-03-08
+47 -30
View File
@@ -97,14 +97,28 @@ function hasCommand(cmd: string): boolean {
}
/**
* Extract origin from URL
* Build llms.txt candidates scoped to the requested URL
*/
function getOrigin(url: string): string {
function buildLlmEndpointCandidates(url: string): string[] {
try {
const parsed = new URL(url);
return `${parsed.protocol}//${parsed.host}`;
if (parsed.pathname === "/") {
return [`${parsed.origin}/.well-known/llms.txt`, `${parsed.origin}/llms.txt`, `${parsed.origin}/llms.md`];
}
const trimmedPath = parsed.pathname.replace(/\/+$/, "");
const segments = trimmedPath.split("/").filter(Boolean);
const scopeDepth = parsed.pathname.endsWith("/") ? segments.length : Math.max(segments.length - 1, 1);
const endpoints: string[] = [];
for (let depth = scopeDepth; depth >= 1; depth--) {
const scope = `/${segments.slice(0, depth).join("/")}/`;
endpoints.push(`${parsed.origin}${scope}llms.txt`, `${parsed.origin}${scope}llms.md`);
}
return endpoints;
} catch {
return "";
return [];
}
}
@@ -227,10 +241,14 @@ async function tryMdSuffix(url: string, timeout: number, signal?: AbortSignal):
/**
* Try to fetch LLM-friendly endpoints
*/
async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSignal): Promise<string | null> {
const endpoints = [`${origin}/.well-known/llms.txt`, `${origin}/llms.txt`, `${origin}/llms.md`];
async function tryLlmEndpoints(
url: string,
timeout: number,
signal?: AbortSignal,
): Promise<{ content: string; endpoint: string } | null> {
const endpoints = buildLlmEndpointCandidates(url);
if (signal?.aborted) {
if (signal?.aborted || endpoints.length === 0) {
return null;
}
@@ -240,7 +258,7 @@ async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSi
}
const result = await loadPage(endpoint, { timeout: Math.min(timeout, 5), signal });
if (result.ok && result.content.trim().length > 100 && !looksLikeHtml(result.content)) {
return result.content;
return { content: result.content, endpoint };
}
}
return null;
@@ -634,7 +652,6 @@ async function renderUrl(
// Step 0: Normalize URL (ensure scheme for special handlers)
url = normalizeUrl(url);
const origin = getOrigin(url);
// Step 1: Try special handlers for known sites (unless raw mode)
if (!raw) {
@@ -912,24 +929,7 @@ async function renderUrl(
};
}
// 5C: LLM-friendly endpoints
const llmContent = await tryLlmEndpoints(origin, timeout, signal);
if (llmContent) {
notes.push("Found llms.txt");
const output = finalizeOutput(llmContent);
return {
url,
finalUrl,
contentType: "text/plain",
method: "llms.txt",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
// 5D: Content negotiation
// 5C: Content negotiation
const negotiated = await tryContentNegotiation(url, timeout, signal);
if (negotiated) {
notes.push(`Content negotiation returned ${negotiated.type}`);
@@ -946,7 +946,7 @@ async function renderUrl(
};
}
// 5E: Check for feed alternates
// 5D: Check for feed alternates
const feedAlternates = alternates.filter(alt => !alt.endsWith(".md") && !alt.includes("markdown"));
for (const altUrl of feedAlternates.slice(0, 2)) {
const resolved = altUrl.startsWith("http") ? altUrl : new URL(altUrl, finalUrl).href;
@@ -972,7 +972,7 @@ async function renderUrl(
throw new ToolAbortError();
}
// Step 6: Render HTML with lynx or html2text
// 5E: Render HTML with lynx or html2text
const htmlResult = await renderHtmlToText(finalUrl, rawContent, timeout, useKagiSummarizer, signal);
if (!htmlResult.ok) {
notes.push("html rendering failed (lynx/html2text unavailable)");
@@ -989,7 +989,7 @@ async function renderUrl(
};
}
// Step 7: If lynx output is low quality, try extracting document links
// Step 6: If rendered output is low quality, try more targeted fallbacks
if (isLowQualityOutput(htmlResult.content)) {
const docLinks = extractDocumentLinks(rawContent, finalUrl);
if (docLinks.length > 0) {
@@ -1019,6 +1019,23 @@ async function renderUrl(
notes.push(`Binary fetch failed: ${binary.error}`);
}
}
const llmResult = await tryLlmEndpoints(finalUrl, timeout, signal);
if (llmResult) {
notes.push(`Used llms.txt fallback: ${llmResult.endpoint}`);
const output = finalizeOutput(llmResult.content);
return {
url,
finalUrl,
contentType: "text/plain",
method: "llms.txt",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
notes.push("Page appears to require JavaScript or is mostly navigation");
}
@@ -11,7 +11,8 @@ import * as kagi from "@oh-my-pi/pi-coding-agent/web/kagi";
import type { LoadPageResult } from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils";
import { Snowflake } from "@oh-my-pi/pi-utils";
import * as natives from "@oh-my-pi/pi-natives";
import { ptree, Snowflake } from "@oh-my-pi/pi-utils";
describe("fetch tool Kagi summarization toggle", () => {
let testDir: string;
@@ -375,4 +376,154 @@ describe("fetch tool Kagi summarization toggle", () => {
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain("<html><body>gateway error</body></html>");
});
it("prefers rendered page content over site-wide llms.txt for deep pages", async () => {
const session = createSession({ "fetch.useKagiSummarizer": false });
const tool = new FetchTool(session);
const pageUrl = "https://bun.com/reference/bun/UnixSocketOptions";
const pageHtml = "<html><body><main><h1>UnixSocketOptions</h1><p>Page-specific docs.</p></main></body></html>";
const renderedMarkdown = `# UnixSocketOptions\n\n${"Page-specific API docs. ".repeat(8)}`;
const originalWhich = Bun.which;
Bun.which = (() => null) as typeof Bun.which;
try {
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => {
if (requestedUrl === pageUrl) {
return {
ok: true,
status: 200,
contentType: "text/html",
finalUrl: pageUrl,
content: pageHtml,
};
}
if (requestedUrl === `${pageUrl}.md`) {
return {
ok: false,
status: 404,
contentType: "text/plain",
finalUrl: requestedUrl,
content: "",
};
}
if (requestedUrl === "https://bun.com/llms.txt") {
return {
ok: true,
status: 200,
contentType: "text/plain",
finalUrl: requestedUrl,
content: `# Bun\n\n${"Site-wide overview. ".repeat(12)}`,
};
}
return {
ok: false,
status: 404,
contentType: "text/plain",
finalUrl: requestedUrl,
content: "",
};
});
vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" }));
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue(undefined);
vi.spyOn(natives, "htmlToMarkdown").mockResolvedValue(renderedMarkdown);
const result = await tool.execute("fetch-deep-page", { url: pageUrl });
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
const textBlock = result.content.find(content => content.type === "text");
expect(result.details?.method).toBe("native");
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain("UnixSocketOptions");
expect(requestedUrls).not.toContain("https://bun.com/.well-known/llms.txt");
expect(requestedUrls).not.toContain("https://bun.com/llms.txt");
expect(requestedUrls).not.toContain("https://bun.com/llms.md");
} finally {
Bun.which = originalWhich;
}
});
it("uses section-scoped llms.txt fallback without requesting the site-wide file", async () => {
const session = createSession({ "fetch.useKagiSummarizer": false });
const tool = new FetchTool(session);
const pageUrl = "https://example.com/docs/reference/widget";
const pageHtml = "<html><body><nav>Docs</nav><main><h1>Widget</h1></main></body></html>";
const lowQualityRender = `${"Please enable JavaScript to view this page.\n".repeat(6)}${"navigation\n".repeat(4)}`;
const originalWhich = Bun.which;
const execSpy = vi.spyOn(ptree, "exec").mockResolvedValue({ ok: true, stdout: lowQualityRender } as never);
Bun.which = (() => null) as typeof Bun.which;
try {
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => {
if (requestedUrl === pageUrl) {
return {
ok: true,
status: 200,
contentType: "text/html",
finalUrl: pageUrl,
content: pageHtml,
};
}
if ([`${pageUrl}.md`, "https://example.com/docs/reference/llms.txt", "https://example.com/docs/reference/llms.md", "https://example.com/docs/llms.md"].includes(requestedUrl)) {
return {
ok: false,
status: 404,
contentType: "text/plain",
finalUrl: requestedUrl,
content: "",
};
}
if (requestedUrl === "https://example.com/docs/llms.txt") {
return {
ok: true,
status: 200,
contentType: "text/plain",
finalUrl: requestedUrl,
content: `# Example Docs\n\n${"Section-scoped fallback. ".repeat(10)}`,
};
}
if (requestedUrl === "https://example.com/llms.txt") {
return {
ok: true,
status: 200,
contentType: "text/plain",
finalUrl: requestedUrl,
content: `# Example\n\n${"Site-wide fallback. ".repeat(10)}`,
};
}
return {
ok: false,
status: 404,
contentType: "text/plain",
finalUrl: requestedUrl,
content: "",
};
});
vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" }));
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue("/usr/bin/trafilatura");
const result = await tool.execute("fetch-section-llms", { url: pageUrl });
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
const textBlock = result.content.find(content => content.type === "text");
expect(result.details?.method).toBe("llms.txt");
expect(result.details?.notes).toContain("Used llms.txt fallback: https://example.com/docs/llms.txt");
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain("Section-scoped fallback");
expect(requestedUrls).toContain("https://example.com/docs/llms.txt");
expect(requestedUrls).not.toContain("https://example.com/.well-known/llms.txt");
expect(requestedUrls).not.toContain("https://example.com/llms.txt");
expect(requestedUrls).not.toContain("https://example.com/llms.md");
} finally {
Bun.which = originalWhich;
execSpy.mockRestore();
}
});
});