feat(coding-agent/tools): implemented path-scoped llms.txt discovery with content-quality fallback
- Moved llms.txt endpoint discovery to fallback strategy when rendered page content is low quality, prioritizing page-specific content over site-wide files. - Enhanced llms.txt endpoint detection to scope candidates to the requested URL path, searching section-specific files before site-wide ones. - Replaced getOrigin() with buildLlmEndpointCandidates() to generate path-scoped endpoint candidates with depth-based fallback strategy. - Updated tryLlmEndpoints() to accept full URL and return endpoint metadata alongside content for better fallback tracking. - Added 2 integration tests validating section-scoped llms.txt discovery and preference for rendered content over site-wide files.
This commit is contained in:
@@ -1,6 +1,10 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Changed
|
||||
|
||||
- Moved llms.txt endpoint discovery to fallback strategy when rendered page content is low quality, prioritizing page-specific content over site-wide files
|
||||
- Enhanced llms.txt endpoint detection to scope candidates to the requested URL path, searching section-specific files before site-wide ones
|
||||
|
||||
## [13.9.6] - 2026-03-08
|
||||
|
||||
|
||||
@@ -97,14 +97,28 @@ function hasCommand(cmd: string): boolean {
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract origin from URL
|
||||
* Build llms.txt candidates scoped to the requested URL
|
||||
*/
|
||||
function getOrigin(url: string): string {
|
||||
function buildLlmEndpointCandidates(url: string): string[] {
|
||||
try {
|
||||
const parsed = new URL(url);
|
||||
return `${parsed.protocol}//${parsed.host}`;
|
||||
if (parsed.pathname === "/") {
|
||||
return [`${parsed.origin}/.well-known/llms.txt`, `${parsed.origin}/llms.txt`, `${parsed.origin}/llms.md`];
|
||||
}
|
||||
|
||||
const trimmedPath = parsed.pathname.replace(/\/+$/, "");
|
||||
const segments = trimmedPath.split("/").filter(Boolean);
|
||||
const scopeDepth = parsed.pathname.endsWith("/") ? segments.length : Math.max(segments.length - 1, 1);
|
||||
const endpoints: string[] = [];
|
||||
|
||||
for (let depth = scopeDepth; depth >= 1; depth--) {
|
||||
const scope = `/${segments.slice(0, depth).join("/")}/`;
|
||||
endpoints.push(`${parsed.origin}${scope}llms.txt`, `${parsed.origin}${scope}llms.md`);
|
||||
}
|
||||
|
||||
return endpoints;
|
||||
} catch {
|
||||
return "";
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -227,10 +241,14 @@ async function tryMdSuffix(url: string, timeout: number, signal?: AbortSignal):
|
||||
/**
|
||||
* Try to fetch LLM-friendly endpoints
|
||||
*/
|
||||
async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSignal): Promise<string | null> {
|
||||
const endpoints = [`${origin}/.well-known/llms.txt`, `${origin}/llms.txt`, `${origin}/llms.md`];
|
||||
async function tryLlmEndpoints(
|
||||
url: string,
|
||||
timeout: number,
|
||||
signal?: AbortSignal,
|
||||
): Promise<{ content: string; endpoint: string } | null> {
|
||||
const endpoints = buildLlmEndpointCandidates(url);
|
||||
|
||||
if (signal?.aborted) {
|
||||
if (signal?.aborted || endpoints.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -240,7 +258,7 @@ async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSi
|
||||
}
|
||||
const result = await loadPage(endpoint, { timeout: Math.min(timeout, 5), signal });
|
||||
if (result.ok && result.content.trim().length > 100 && !looksLikeHtml(result.content)) {
|
||||
return result.content;
|
||||
return { content: result.content, endpoint };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
@@ -634,7 +652,6 @@ async function renderUrl(
|
||||
|
||||
// Step 0: Normalize URL (ensure scheme for special handlers)
|
||||
url = normalizeUrl(url);
|
||||
const origin = getOrigin(url);
|
||||
|
||||
// Step 1: Try special handlers for known sites (unless raw mode)
|
||||
if (!raw) {
|
||||
@@ -912,24 +929,7 @@ async function renderUrl(
|
||||
};
|
||||
}
|
||||
|
||||
// 5C: LLM-friendly endpoints
|
||||
const llmContent = await tryLlmEndpoints(origin, timeout, signal);
|
||||
if (llmContent) {
|
||||
notes.push("Found llms.txt");
|
||||
const output = finalizeOutput(llmContent);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: "text/plain",
|
||||
method: "llms.txt",
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
// 5D: Content negotiation
|
||||
// 5C: Content negotiation
|
||||
const negotiated = await tryContentNegotiation(url, timeout, signal);
|
||||
if (negotiated) {
|
||||
notes.push(`Content negotiation returned ${negotiated.type}`);
|
||||
@@ -946,7 +946,7 @@ async function renderUrl(
|
||||
};
|
||||
}
|
||||
|
||||
// 5E: Check for feed alternates
|
||||
// 5D: Check for feed alternates
|
||||
const feedAlternates = alternates.filter(alt => !alt.endsWith(".md") && !alt.includes("markdown"));
|
||||
for (const altUrl of feedAlternates.slice(0, 2)) {
|
||||
const resolved = altUrl.startsWith("http") ? altUrl : new URL(altUrl, finalUrl).href;
|
||||
@@ -972,7 +972,7 @@ async function renderUrl(
|
||||
throw new ToolAbortError();
|
||||
}
|
||||
|
||||
// Step 6: Render HTML with lynx or html2text
|
||||
// 5E: Render HTML with lynx or html2text
|
||||
const htmlResult = await renderHtmlToText(finalUrl, rawContent, timeout, useKagiSummarizer, signal);
|
||||
if (!htmlResult.ok) {
|
||||
notes.push("html rendering failed (lynx/html2text unavailable)");
|
||||
@@ -989,7 +989,7 @@ async function renderUrl(
|
||||
};
|
||||
}
|
||||
|
||||
// Step 7: If lynx output is low quality, try extracting document links
|
||||
// Step 6: If rendered output is low quality, try more targeted fallbacks
|
||||
if (isLowQualityOutput(htmlResult.content)) {
|
||||
const docLinks = extractDocumentLinks(rawContent, finalUrl);
|
||||
if (docLinks.length > 0) {
|
||||
@@ -1019,6 +1019,23 @@ async function renderUrl(
|
||||
notes.push(`Binary fetch failed: ${binary.error}`);
|
||||
}
|
||||
}
|
||||
|
||||
const llmResult = await tryLlmEndpoints(finalUrl, timeout, signal);
|
||||
if (llmResult) {
|
||||
notes.push(`Used llms.txt fallback: ${llmResult.endpoint}`);
|
||||
const output = finalizeOutput(llmResult.content);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: "text/plain",
|
||||
method: "llms.txt",
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
notes.push("Page appears to require JavaScript or is mostly navigation");
|
||||
}
|
||||
|
||||
|
||||
@@ -11,7 +11,8 @@ import * as kagi from "@oh-my-pi/pi-coding-agent/web/kagi";
|
||||
import type { LoadPageResult } from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils";
|
||||
import { Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import * as natives from "@oh-my-pi/pi-natives";
|
||||
import { ptree, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
|
||||
describe("fetch tool Kagi summarization toggle", () => {
|
||||
let testDir: string;
|
||||
@@ -375,4 +376,154 @@ describe("fetch tool Kagi summarization toggle", () => {
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain("<html><body>gateway error</body></html>");
|
||||
});
|
||||
it("prefers rendered page content over site-wide llms.txt for deep pages", async () => {
|
||||
const session = createSession({ "fetch.useKagiSummarizer": false });
|
||||
const tool = new FetchTool(session);
|
||||
const pageUrl = "https://bun.com/reference/bun/UnixSocketOptions";
|
||||
const pageHtml = "<html><body><main><h1>UnixSocketOptions</h1><p>Page-specific docs.</p></main></body></html>";
|
||||
const renderedMarkdown = `# UnixSocketOptions\n\n${"Page-specific API docs. ".repeat(8)}`;
|
||||
const originalWhich = Bun.which;
|
||||
|
||||
Bun.which = (() => null) as typeof Bun.which;
|
||||
try {
|
||||
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => {
|
||||
if (requestedUrl === pageUrl) {
|
||||
return {
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/html",
|
||||
finalUrl: pageUrl,
|
||||
content: pageHtml,
|
||||
};
|
||||
}
|
||||
|
||||
if (requestedUrl === `${pageUrl}.md`) {
|
||||
return {
|
||||
ok: false,
|
||||
status: 404,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: "",
|
||||
};
|
||||
}
|
||||
|
||||
if (requestedUrl === "https://bun.com/llms.txt") {
|
||||
return {
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: `# Bun\n\n${"Site-wide overview. ".repeat(12)}`,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
ok: false,
|
||||
status: 404,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: "",
|
||||
};
|
||||
});
|
||||
vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" }));
|
||||
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue(undefined);
|
||||
vi.spyOn(natives, "htmlToMarkdown").mockResolvedValue(renderedMarkdown);
|
||||
|
||||
const result = await tool.execute("fetch-deep-page", { url: pageUrl });
|
||||
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
expect(result.details?.method).toBe("native");
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain("UnixSocketOptions");
|
||||
expect(requestedUrls).not.toContain("https://bun.com/.well-known/llms.txt");
|
||||
expect(requestedUrls).not.toContain("https://bun.com/llms.txt");
|
||||
expect(requestedUrls).not.toContain("https://bun.com/llms.md");
|
||||
} finally {
|
||||
Bun.which = originalWhich;
|
||||
}
|
||||
});
|
||||
|
||||
it("uses section-scoped llms.txt fallback without requesting the site-wide file", async () => {
|
||||
const session = createSession({ "fetch.useKagiSummarizer": false });
|
||||
const tool = new FetchTool(session);
|
||||
const pageUrl = "https://example.com/docs/reference/widget";
|
||||
const pageHtml = "<html><body><nav>Docs</nav><main><h1>Widget</h1></main></body></html>";
|
||||
const lowQualityRender = `${"Please enable JavaScript to view this page.\n".repeat(6)}${"navigation\n".repeat(4)}`;
|
||||
const originalWhich = Bun.which;
|
||||
const execSpy = vi.spyOn(ptree, "exec").mockResolvedValue({ ok: true, stdout: lowQualityRender } as never);
|
||||
|
||||
Bun.which = (() => null) as typeof Bun.which;
|
||||
try {
|
||||
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockImplementation(async (requestedUrl: string) => {
|
||||
if (requestedUrl === pageUrl) {
|
||||
return {
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/html",
|
||||
finalUrl: pageUrl,
|
||||
content: pageHtml,
|
||||
};
|
||||
}
|
||||
|
||||
if ([`${pageUrl}.md`, "https://example.com/docs/reference/llms.txt", "https://example.com/docs/reference/llms.md", "https://example.com/docs/llms.md"].includes(requestedUrl)) {
|
||||
return {
|
||||
ok: false,
|
||||
status: 404,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: "",
|
||||
};
|
||||
}
|
||||
|
||||
if (requestedUrl === "https://example.com/docs/llms.txt") {
|
||||
return {
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: `# Example Docs\n\n${"Section-scoped fallback. ".repeat(10)}`,
|
||||
};
|
||||
}
|
||||
|
||||
if (requestedUrl === "https://example.com/llms.txt") {
|
||||
return {
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: `# Example\n\n${"Site-wide fallback. ".repeat(10)}`,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
ok: false,
|
||||
status: 404,
|
||||
contentType: "text/plain",
|
||||
finalUrl: requestedUrl,
|
||||
content: "",
|
||||
};
|
||||
});
|
||||
vi.spyOn(globalThis, "fetch").mockResolvedValue(new Response("blocked", { status: 500, statusText: "Blocked" }));
|
||||
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue("/usr/bin/trafilatura");
|
||||
|
||||
const result = await tool.execute("fetch-section-llms", { url: pageUrl });
|
||||
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
expect(result.details?.method).toBe("llms.txt");
|
||||
expect(result.details?.notes).toContain("Used llms.txt fallback: https://example.com/docs/llms.txt");
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain("Section-scoped fallback");
|
||||
expect(requestedUrls).toContain("https://example.com/docs/llms.txt");
|
||||
expect(requestedUrls).not.toContain("https://example.com/.well-known/llms.txt");
|
||||
expect(requestedUrls).not.toContain("https://example.com/llms.txt");
|
||||
expect(requestedUrls).not.toContain("https://example.com/llms.md");
|
||||
} finally {
|
||||
Bun.which = originalWhich;
|
||||
execSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user