feat(web-scrapers): added 25+ specialized scrapers with modular architecture

- Added 25+ new specialized web scrapers for sites including MusicBrainz, Discourse, Lemmy, Sourcegraph, ORCID, and package registries.
- Refactored web-fetch-handlers module to web-scrapers with modular architecture and centralized handler exports.
- Added AbortSignal support throughout web-fetch tool and scraper utilities for improved request cancellation.
- Fixed npm license parsing to handle object format with type property.
- Fixed raw GitHub URL construction by removing refs/heads prefix from ref path.
This commit is contained in:
can1357
2026-01-07 20:34:33 +01:00
parent d5530d428b
commit 5a7cf77ba9
99 changed files with 4654 additions and 566 deletions
+17
View File
@@ -1,6 +1,23 @@
# Changelog
## [Unreleased]
### Added
- Added 80+ specialized web scrapers for structured content extraction from popular sites including GitHub, GitLab, npm, PyPI, crates.io, Wikipedia, YouTube, Stack Overflow, Hacker News, Reddit, arXiv, PubMed, and many more
- Added site-specific API integrations for package registries (npm, PyPI, crates.io, Hex, Hackage, NuGet, Maven, RubyGems, Packagist, pub.dev, Go packages)
- Added scrapers for social platforms (Mastodon, Bluesky, Lemmy, Lobsters, Dev.to, Discourse)
- Added scrapers for academic sources (arXiv, bioRxiv, PubMed, Semantic Scholar, ORCID, CrossRef, IACR)
- Added scrapers for security databases (NVD, OSV, CISA KEV)
- Added scrapers for documentation sites (MDN, Read the Docs, RFC Editor, W3C, SPDX, tldr, cheat.sh)
- Added scrapers for media platforms (YouTube, Vimeo, Spotify, Discogs, MusicBrainz)
- Added scrapers for AI/ML platforms (Hugging Face, Ollama)
- Added scrapers for app stores and marketplaces (VS Code Marketplace, JetBrains Marketplace, Firefox Add-ons, Open VSX, Flathub, F-Droid, Snapcraft)
- Added scrapers for business data (SEC EDGAR, OpenCorporates, CoinGecko)
- Added scrapers for reference sources (Wikipedia, Wikidata, OpenLibrary, Choose a License)
### Changed
- Refactored web-fetch tool to use modular scraper architecture for improved maintainability
## [3.30.0] - 2026-01-07
### Added
@@ -1,69 +0,0 @@
/**
* Web Fetch Special Handlers Index
*
* Exports all special handlers for site-specific content extraction.
*/
export { handleArtifactHub } from "./artifacthub";
// Academic
export { handleArxiv } from "./arxiv";
export { handleAur } from "./aur";
export { handleBiorxiv } from "./biorxiv";
export { handleBluesky } from "./bluesky";
export { handleBrew } from "./brew";
export { handleCheatSh } from "./cheatsh";
export { handleChocolatey } from "./chocolatey";
export { handleCoinGecko } from "./coingecko";
export { handleCratesIo } from "./crates-io";
export { handleDevTo } from "./devto";
export { handleDiscogs } from "./discogs";
export { handleDockerHub } from "./dockerhub";
// Git hosting
export { fetchGitHubApi, handleGitHub } from "./github";
export { handleGitHubGist } from "./github-gist";
export { handleGitLab } from "./gitlab";
export { handleGoPkg } from "./go-pkg";
export { handleHackage } from "./hackage";
export { handleHackerNews } from "./hackernews";
export { handleHex } from "./hex";
// ML/AI
export { handleHuggingFace } from "./huggingface";
export { handleIacr } from "./iacr";
export { handleLobsters } from "./lobsters";
export { handleMastodon } from "./mastodon";
export { handleMaven } from "./maven";
export { handleMDN } from "./mdn";
export { handleMetaCPAN } from "./metacpan";
// Package registries
export { handleNpm } from "./npm";
export { handleNuGet } from "./nuget";
export { handleNvd } from "./nvd";
export { handleOpenCorporates } from "./opencorporates";
export { handleOpenLibrary } from "./openlibrary";
export { handleOsv } from "./osv";
export { handlePackagist } from "./packagist";
export { handlePubDev } from "./pub-dev";
export { handlePubMed } from "./pubmed";
export { handlePyPI } from "./pypi";
export { handleReadTheDocs } from "./readthedocs";
export { handleReddit } from "./reddit";
export { handleRepology } from "./repology";
export { handleRfc } from "./rfc";
export { handleRubyGems } from "./rubygems";
export { handleSecEdgar } from "./sec-edgar";
export { handleSemanticScholar } from "./semantic-scholar";
export { handleSpotify } from "./spotify";
// Developer content
export { handleStackOverflow } from "./stackoverflow";
export { handleTerraform } from "./terraform";
export { handleTldr } from "./tldr";
// Social/News
export { handleTwitter } from "./twitter";
export type { RenderResult, SpecialHandler } from "./types";
export { handleVimeo } from "./vimeo";
// Reference
export { handleWikidata } from "./wikidata";
export { handleWikipedia } from "./wikipedia";
// Video/Media
export { handleYouTube } from "./youtube";
@@ -1,91 +0,0 @@
import { tmpdir } from "node:os";
import * as path from "node:path";
import { ensureTool } from "../../../utils/tools-manager";
const MAX_BYTES = 50 * 1024 * 1024; // 50MB for binary files
function exec(
cmd: string,
args: string[],
options?: { timeout?: number; input?: string | Buffer },
): { stdout: string; stderr: string; ok: boolean } {
const result = Bun.spawnSync([cmd, ...args], {
stdin: options?.input ? (options.input as any) : "ignore",
stdout: "pipe",
stderr: "pipe",
});
return {
stdout: result.stdout?.toString() ?? "",
stderr: result.stderr?.toString() ?? "",
ok: result.exitCode === 0,
};
}
export async function convertWithMarkitdown(
content: Buffer,
extensionHint: string,
timeout: number,
): Promise<{ content: string; ok: boolean }> {
const markitdown = await ensureTool("markitdown", true);
if (!markitdown) {
return { content: "", ok: false };
}
// Write to temp file with extension hint
const ext = extensionHint || ".bin";
const tmpDir = tmpdir();
const tmpFile = path.join(tmpDir, `omp-convert-${Date.now()}${ext}`);
try {
await Bun.write(tmpFile, content);
const result = exec(markitdown, [tmpFile], { timeout });
return { content: result.stdout, ok: result.ok };
} finally {
try {
await Bun.$`rm ${tmpFile}`.quiet();
} catch {}
}
}
export async function fetchBinary(
url: string,
timeout: number,
): Promise<{ buffer: Buffer; contentType: string; contentDisposition?: string; ok: boolean }> {
try {
const controller = new AbortController();
const timeoutId = setTimeout(() => controller.abort(), timeout * 1000);
const response = await fetch(url, {
signal: controller.signal,
headers: {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0",
},
redirect: "follow",
});
clearTimeout(timeoutId);
if (!response.ok) {
return { buffer: Buffer.alloc(0), contentType: "", ok: false };
}
const contentType = response.headers.get("content-type") ?? "";
const contentDisposition = response.headers.get("content-disposition") ?? undefined;
const contentLength = response.headers.get("content-length");
if (contentLength) {
const size = Number.parseInt(contentLength, 10);
if (Number.isFinite(size) && size > MAX_BYTES) {
return { buffer: Buffer.alloc(0), contentType, contentDisposition, ok: false };
}
}
const buffer = Buffer.from(await response.arrayBuffer());
if (buffer.length > MAX_BYTES) {
return { buffer: Buffer.alloc(0), contentType, contentDisposition, ok: false };
}
return { buffer, contentType, contentDisposition, ok: true };
} catch {
return { buffer: Buffer.alloc(0), contentType: "", ok: false };
}
}
+106 -371
View File
@@ -5,82 +5,17 @@ import { Type } from "@sinclair/typebox";
import { parse as parseHtml } from "node-html-parser";
import webFetchDescription from "../../prompts/tools/web-fetch.md" with { type: "text" };
import { ensureTool } from "../../utils/tools-manager";
import { logger } from "../logger";
import type { ToolSession } from "./index";
import {
handleArtifactHub,
handleArxiv,
handleAur,
handleBiorxiv,
handleBluesky,
handleBrew,
handleCheatSh,
handleChocolatey,
handleCoinGecko,
handleCratesIo,
handleDevTo,
handleDiscogs,
handleDockerHub,
handleGitHub,
handleGitHubGist,
handleGitLab,
handleGoPkg,
handleHackage,
handleHackerNews,
handleHex,
handleHuggingFace,
handleIacr,
handleLobsters,
handleMastodon,
handleMaven,
handleMDN,
handleMetaCPAN,
handleNpm,
handleNuGet,
handleNvd,
handleOpenCorporates,
handleOpenLibrary,
handleOsv,
handlePackagist,
handlePubDev,
handlePubMed,
handlePyPI,
handleReadTheDocs,
handleReddit,
handleRepology,
handleRfc,
handleRubyGems,
handleSecEdgar,
handleSemanticScholar,
handleSpotify,
handleStackOverflow,
handleTerraform,
handleTldr,
handleTwitter,
handleVimeo,
handleWikidata,
handleWikipedia,
handleYouTube,
} from "./web-fetch-handlers/index";
import { specialHandlers } from "./web-scrapers/index";
import type { RenderResult } from "./web-scrapers/types";
import { finalizeOutput, loadPage } from "./web-scrapers/types";
import { convertWithMarkitdown, fetchBinary } from "./web-scrapers/utils";
// =============================================================================
// Types and Constants
// =============================================================================
interface RenderResult {
url: string;
finalUrl: string;
contentType: string;
method: string;
content: string;
fetchedAt: string;
truncated: boolean;
notes: string[];
}
const DEFAULT_TIMEOUT = 20;
const MAX_BYTES = 50 * 1024 * 1024; // 50MB for binary files
const MAX_OUTPUT_CHARS = 500_000;
// Convertible document types (markitdown supported)
const CONVERTIBLE_MIMES = new Set([
@@ -123,124 +58,11 @@ const CONVERTIBLE_EXTENSIONS = new Set([
".ogg",
]);
const USER_AGENTS = [
"curl/8.0",
"Mozilla/5.0 (compatible; TextBot/1.0)",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
];
// =============================================================================
// Utilities
// =============================================================================
interface LoadPageResult {
content: string;
contentType: string;
finalUrl: string;
ok: boolean;
status?: number;
}
interface LoadPageOptions {
timeout?: number;
headers?: Record<string, string>;
maxBytes?: number;
}
/**
* Check if response indicates bot blocking (Cloudflare, etc.)
*/
function isBotBlocked(status: number, content: string): boolean {
if (status === 403 || status === 503) {
const lower = content.toLowerCase();
return (
lower.includes("cloudflare") ||
lower.includes("captcha") ||
lower.includes("challenge") ||
lower.includes("blocked") ||
lower.includes("access denied") ||
lower.includes("bot detection")
);
}
return false;
}
/**
* Fetch a page with timeout, size limit, and automatic retry with browser UA if blocked
*/
async function loadPage(url: string, options: LoadPageOptions = {}): Promise<LoadPageResult> {
const { timeout = 20, headers = {}, maxBytes = MAX_BYTES } = options;
for (let attempt = 0; attempt < USER_AGENTS.length; attempt++) {
const userAgent = USER_AGENTS[attempt];
try {
const controller = new AbortController();
const timeoutId = setTimeout(() => controller.abort(), timeout * 1000);
const response = await fetch(url, {
signal: controller.signal,
headers: {
"User-Agent": userAgent,
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.5",
...headers,
},
redirect: "follow",
});
clearTimeout(timeoutId);
const contentType = response.headers.get("content-type")?.split(";")[0]?.trim().toLowerCase() ?? "";
const finalUrl = response.url;
// Read with size limit
const reader = response.body?.getReader();
if (!reader) {
return { content: "", contentType, finalUrl, ok: false, status: response.status };
}
const chunks: Uint8Array[] = [];
let totalSize = 0;
while (true) {
const { done, value } = await reader.read();
if (done) break;
chunks.push(value);
totalSize += value.length;
if (totalSize > maxBytes) {
reader.cancel();
break;
}
}
const decoder = new TextDecoder();
const content = decoder.decode(Buffer.concat(chunks));
// Check if we got blocked and should retry with browser UA
if (isBotBlocked(response.status, content) && attempt < USER_AGENTS.length - 1) {
continue;
}
if (!response.ok) {
return { content, contentType, finalUrl, ok: false, status: response.status };
}
return { content, contentType, finalUrl, ok: true, status: response.status };
} catch (err) {
// On last attempt, return failure
if (attempt === USER_AGENTS.length - 1) {
logger.debug("Web fetch failed after retries", { url, error: String(err) });
return { content: "", contentType: "", finalUrl: url, ok: false };
}
// Otherwise retry with next UA
}
}
return { content: "", contentType: "", finalUrl: url, ok: false };
}
type SpawnSyncOptions = NonNullable<Parameters<typeof Bun.spawnSync>[1]>;
/**
* Execute a command and return stdout
@@ -250,8 +72,9 @@ function exec(
args: string[],
options?: { timeout?: number; input?: string | Buffer },
): { stdout: string; stderr: string; ok: boolean } {
const stdin = (options?.input ?? "ignore") as SpawnSyncOptions["stdin"];
const result = Bun.spawnSync([cmd, ...args], {
stdin: options?.input ? (options.input as any) : "ignore",
stdin,
stdout: "pipe",
stderr: "pipe",
});
@@ -344,39 +167,10 @@ function looksLikeHtml(content: string): boolean {
);
}
/**
* Convert binary file to markdown using markitdown
*/
async function convertWithMarkitdown(
content: Buffer,
extensionHint: string,
timeout: number,
): Promise<{ content: string; ok: boolean }> {
const markitdown = await ensureTool("markitdown", true);
if (!markitdown) {
return { content: "", ok: false };
}
// Write to temp file with extension hint
const ext = extensionHint || ".bin";
const tmpDir = tmpdir();
const tmpFile = path.join(tmpDir, `omp-convert-${Date.now()}${ext}`);
try {
await Bun.write(tmpFile, content);
const result = exec(markitdown, [tmpFile], { timeout });
return { content: result.stdout, ok: result.ok };
} finally {
try {
await Bun.$`rm ${tmpFile}`.quiet();
} catch {}
}
}
/**
* Try fetching URL with .md appended (llms.txt convention)
*/
async function tryMdSuffix(url: string, timeout: number): Promise<string | null> {
async function tryMdSuffix(url: string, timeout: number, signal?: AbortSignal): Promise<string | null> {
const candidates: string[] = [];
try {
@@ -397,8 +191,15 @@ async function tryMdSuffix(url: string, timeout: number): Promise<string | null>
return null;
}
if (signal?.aborted) {
return null;
}
for (const candidate of candidates) {
const result = await loadPage(candidate, { timeout: Math.min(timeout, 5) });
if (signal?.aborted) {
return null;
}
const result = await loadPage(candidate, { timeout: Math.min(timeout, 5), signal });
if (result.ok && result.content.trim().length > 100 && !looksLikeHtml(result.content)) {
return result.content;
}
@@ -410,11 +211,18 @@ async function tryMdSuffix(url: string, timeout: number): Promise<string | null>
/**
* Try to fetch LLM-friendly endpoints
*/
async function tryLlmEndpoints(origin: string, timeout: number): Promise<string | null> {
async function tryLlmEndpoints(origin: string, timeout: number, signal?: AbortSignal): Promise<string | null> {
const endpoints = [`${origin}/.well-known/llms.txt`, `${origin}/llms.txt`, `${origin}/llms.md`];
if (signal?.aborted) {
return null;
}
for (const endpoint of endpoints) {
const result = await loadPage(endpoint, { timeout: Math.min(timeout, 5) });
if (signal?.aborted) {
return null;
}
const result = await loadPage(endpoint, { timeout: Math.min(timeout, 5), signal });
if (result.ok && result.content.trim().length > 100 && !looksLikeHtml(result.content)) {
return result.content;
}
@@ -425,10 +233,19 @@ async function tryLlmEndpoints(origin: string, timeout: number): Promise<string
/**
* Try content negotiation for markdown/plain
*/
async function tryContentNegotiation(url: string, timeout: number): Promise<{ content: string; type: string } | null> {
async function tryContentNegotiation(
url: string,
timeout: number,
signal?: AbortSignal,
): Promise<{ content: string; type: string } | null> {
if (signal?.aborted) {
return null;
}
const result = await loadPage(url, {
timeout,
headers: { Accept: "text/markdown, text/plain;q=0.9, text/html;q=0.8" },
signal,
});
if (!result.ok) return null;
@@ -658,64 +475,6 @@ function formatJson(content: string): string {
}
}
/**
* Truncate and cleanup output
*/
function finalizeOutput(content: string): { content: string; truncated: boolean } {
const cleaned = content.replace(/\n{3,}/g, "\n\n").trim();
const truncated = cleaned.length > MAX_OUTPUT_CHARS;
return {
content: cleaned.slice(0, MAX_OUTPUT_CHARS),
truncated,
};
}
/**
* Fetch page as binary buffer (for convertible files)
*/
async function fetchBinary(
url: string,
timeout: number,
): Promise<{ buffer: Buffer; contentType: string; contentDisposition?: string; ok: boolean }> {
try {
const controller = new AbortController();
const timeoutId = setTimeout(() => controller.abort(), timeout * 1000);
const response = await fetch(url, {
signal: controller.signal,
headers: {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0",
},
redirect: "follow",
});
clearTimeout(timeoutId);
if (!response.ok) {
return { buffer: Buffer.alloc(0), contentType: "", ok: false };
}
const contentType = response.headers.get("content-type") ?? "";
const contentDisposition = response.headers.get("content-disposition") ?? undefined;
const contentLength = response.headers.get("content-length");
if (contentLength) {
const size = Number.parseInt(contentLength, 10);
if (Number.isFinite(size) && size > MAX_BYTES) {
return { buffer: Buffer.alloc(0), contentType, contentDisposition, ok: false };
}
}
const buffer = Buffer.from(await response.arrayBuffer());
if (buffer.length > MAX_BYTES) {
return { buffer: Buffer.alloc(0), contentType, contentDisposition, ok: false };
}
return { buffer, contentType, contentDisposition, ok: true };
} catch {
return { buffer: Buffer.alloc(0), contentType: "", ok: false };
}
}
// =============================================================================
// Unified Special Handler Dispatch
// =============================================================================
@@ -723,74 +482,15 @@ async function fetchBinary(
/**
* Try all special handlers
*/
async function handleSpecialUrls(url: string, timeout: number): Promise<RenderResult | null> {
// Order matters - more specific first
return (
// Git hosting
(await handleGitHubGist(url, timeout)) ||
(await handleGitHub(url, timeout)) ||
(await handleGitLab(url, timeout)) ||
// Video/Media
(await handleYouTube(url, timeout)) ||
(await handleVimeo(url, timeout)) ||
(await handleSpotify(url, timeout)) ||
(await handleDiscogs(url, timeout)) ||
// Social/News
(await handleTwitter(url, timeout)) ||
(await handleBluesky(url, timeout)) ||
(await handleMastodon(url, timeout)) ||
(await handleHackerNews(url, timeout)) ||
(await handleLobsters(url, timeout)) ||
(await handleReddit(url, timeout)) ||
// Developer content
(await handleStackOverflow(url, timeout)) ||
(await handleDevTo(url, timeout)) ||
(await handleMDN(url, timeout)) ||
(await handleReadTheDocs(url, timeout)) ||
(await handleTldr(url, timeout)) ||
(await handleCheatSh(url, timeout)) ||
// Package registries
(await handleNpm(url, timeout)) ||
(await handleNuGet(url, timeout)) ||
(await handleChocolatey(url, timeout)) ||
(await handleBrew(url, timeout)) ||
(await handlePyPI(url, timeout)) ||
(await handleCratesIo(url, timeout)) ||
(await handleDockerHub(url, timeout)) ||
(await handleGoPkg(url, timeout)) ||
(await handleHex(url, timeout)) ||
(await handlePackagist(url, timeout)) ||
(await handlePubDev(url, timeout)) ||
(await handleMaven(url, timeout)) ||
(await handleArtifactHub(url, timeout)) ||
(await handleRubyGems(url, timeout)) ||
(await handleTerraform(url, timeout)) ||
(await handleAur(url, timeout)) ||
(await handleHackage(url, timeout)) ||
(await handleMetaCPAN(url, timeout)) ||
(await handleRepology(url, timeout)) ||
// ML/AI
(await handleHuggingFace(url, timeout)) ||
// Academic
(await handleArxiv(url, timeout)) ||
(await handleBiorxiv(url, timeout)) ||
(await handleIacr(url, timeout)) ||
(await handleSemanticScholar(url, timeout)) ||
(await handlePubMed(url, timeout)) ||
(await handleRfc(url, timeout)) ||
// Security
(await handleNvd(url, timeout)) ||
(await handleOsv(url, timeout)) ||
// Crypto
(await handleCoinGecko(url, timeout)) ||
// Business
(await handleOpenCorporates(url, timeout)) ||
(await handleSecEdgar(url, timeout)) ||
// Reference
(await handleOpenLibrary(url, timeout)) ||
(await handleWikidata(url, timeout)) ||
(await handleWikipedia(url, timeout))
);
async function handleSpecialUrls(url: string, timeout: number, signal?: AbortSignal): Promise<RenderResult | null> {
for (const handler of specialHandlers) {
if (signal?.aborted) {
throw new Error("Operation aborted");
}
const result = await handler(url, timeout);
if (result) return result;
}
return null;
}
// =============================================================================
@@ -800,9 +500,17 @@ async function handleSpecialUrls(url: string, timeout: number): Promise<RenderRe
/**
* Main render function implementing the full pipeline
*/
async function renderUrl(url: string, timeout: number, raw: boolean = false): Promise<RenderResult> {
async function renderUrl(
url: string,
timeout: number,
raw: boolean = false,
signal?: AbortSignal,
): Promise<RenderResult> {
const notes: string[] = [];
const fetchedAt = new Date().toISOString();
if (signal?.aborted) {
throw new Error("Operation aborted");
}
// Step 0: Normalize URL (ensure scheme for special handlers)
url = normalizeUrl(url);
@@ -810,12 +518,15 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
// Step 1: Try special handlers for known sites (unless raw mode)
if (!raw) {
const specialResult = await handleSpecialUrls(url, timeout);
const specialResult = await handleSpecialUrls(url, timeout, signal);
if (specialResult) return specialResult;
}
// Step 2: Fetch page
const response = await loadPage(url, { timeout });
const response = await loadPage(url, { timeout, signal });
if (signal?.aborted) {
throw new Error("Operation aborted");
}
if (!response.ok) {
return {
url,
@@ -835,26 +546,36 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
// Step 3: Handle convertible binary files (PDF, DOCX, etc.)
if (isConvertible(mime, extHint)) {
const binary = await fetchBinary(finalUrl, timeout);
const binary = await fetchBinary(finalUrl, timeout, signal);
if (binary.ok) {
const ext = getExtensionHint(finalUrl, binary.contentDisposition) || extHint;
const converted = await convertWithMarkitdown(binary.buffer, ext, timeout);
if (converted.ok && converted.content.trim().length > 50) {
notes.push(`Converted with markitdown`);
const output = finalizeOutput(converted.content);
return {
url,
finalUrl,
contentType: mime,
method: "markitdown",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
const converted = await convertWithMarkitdown(binary.buffer, ext, timeout, signal);
if (converted.ok) {
if (converted.content.trim().length > 50) {
notes.push("Converted with markitdown");
const output = finalizeOutput(converted.content);
return {
url,
finalUrl,
contentType: mime,
method: "markitdown",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
notes.push("markitdown conversion produced no usable output");
} else if (converted.error) {
notes.push(`markitdown conversion failed: ${converted.error}`);
} else {
notes.push("markitdown conversion failed");
}
} else if (binary.error) {
notes.push(`Binary fetch failed: ${binary.error}`);
} else {
notes.push("Binary fetch failed");
}
notes.push("markitdown conversion failed");
}
// Step 4: Handle non-HTML text content
@@ -914,7 +635,7 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
const markdownAlt = alternates.find((alt) => alt.endsWith(".md") || alt.includes("markdown"));
if (markdownAlt) {
const resolved = markdownAlt.startsWith("http") ? markdownAlt : new URL(markdownAlt, finalUrl).href;
const altResult = await loadPage(resolved, { timeout });
const altResult = await loadPage(resolved, { timeout, signal });
if (altResult.ok && altResult.content.trim().length > 100 && !looksLikeHtml(altResult.content)) {
notes.push(`Used markdown alternate: ${resolved}`);
const output = finalizeOutput(altResult.content);
@@ -932,7 +653,7 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
}
// 5B: Try URL.md suffix (llms.txt convention)
const mdSuffix = await tryMdSuffix(finalUrl, timeout);
const mdSuffix = await tryMdSuffix(finalUrl, timeout, signal);
if (mdSuffix) {
notes.push("Found .md suffix version");
const output = finalizeOutput(mdSuffix);
@@ -949,7 +670,7 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
}
// 5C: LLM-friendly endpoints
const llmContent = await tryLlmEndpoints(origin, timeout);
const llmContent = await tryLlmEndpoints(origin, timeout, signal);
if (llmContent) {
notes.push("Found llms.txt");
const output = finalizeOutput(llmContent);
@@ -966,7 +687,7 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
}
// 5D: Content negotiation
const negotiated = await tryContentNegotiation(url, timeout);
const negotiated = await tryContentNegotiation(url, timeout, signal);
if (negotiated) {
notes.push(`Content negotiation returned ${negotiated.type}`);
const output = finalizeOutput(negotiated.content);
@@ -986,7 +707,7 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
const feedAlternates = alternates.filter((alt) => !alt.endsWith(".md") && !alt.includes("markdown"));
for (const altUrl of feedAlternates.slice(0, 2)) {
const resolved = altUrl.startsWith("http") ? altUrl : new URL(altUrl, finalUrl).href;
const altResult = await loadPage(resolved, { timeout });
const altResult = await loadPage(resolved, { timeout, signal });
if (altResult.ok && altResult.content.trim().length > 200) {
notes.push(`Used feed alternate: ${resolved}`);
const parsed = parseFeedToMarkdown(altResult.content);
@@ -1004,6 +725,10 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
}
}
if (signal?.aborted) {
throw new Error("Operation aborted");
}
// Step 6: Render HTML with lynx or html2text
const htmlResult = await renderHtmlToText(rawContent, timeout);
if (!htmlResult.ok) {
@@ -1026,10 +751,10 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
const docLinks = extractDocumentLinks(rawContent, finalUrl);
if (docLinks.length > 0) {
const docUrl = docLinks[0];
const binary = await fetchBinary(docUrl, timeout);
const binary = await fetchBinary(docUrl, timeout, signal);
if (binary.ok) {
const ext = getExtensionHint(docUrl, binary.contentDisposition);
const converted = await convertWithMarkitdown(binary.buffer, ext, timeout);
const converted = await convertWithMarkitdown(binary.buffer, ext, timeout, signal);
if (converted.ok && converted.content.trim().length > htmlResult.content.length) {
notes.push(`Extracted and converted document: ${docUrl}`);
const output = finalizeOutput(converted.content);
@@ -1044,6 +769,11 @@ async function renderUrl(url: string, timeout: number, raw: boolean = false): Pr
notes,
};
}
if (!converted.ok && converted.error) {
notes.push(`markitdown conversion failed: ${converted.error}`);
}
} else if (binary.error) {
notes.push(`Binary fetch failed: ${binary.error}`);
}
}
notes.push("Page appears to require JavaScript or is mostly navigation");
@@ -1106,11 +836,16 @@ export function createWebFetchTool(_session: ToolSession): AgentTool<typeof webF
execute: async (
_toolCallId: string,
{ url, timeout = DEFAULT_TIMEOUT, raw = false }: { url: string; timeout?: number; raw?: boolean },
signal?: AbortSignal,
) => {
if (signal?.aborted) {
throw new Error("Operation aborted");
}
// Clamp timeout
const effectiveTimeout = Math.min(Math.max(timeout, 1), 120);
const result = await renderUrl(url, effectiveTimeout, raw);
const result = await renderUrl(url, effectiveTimeout, raw, signal);
// Format output
let output = "";
@@ -0,0 +1,109 @@
import { parseFrontmatter } from "../../../discovery/helpers";
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
const ALLOWED_HOSTS = new Set(["choosealicense.com", "www.choosealicense.com"]);
const LICENSE_PATH = /^\/licenses\/([^/]+)\/?$/i;
const APPENDIX_PATH = /^\/appendix\/?$/i;
function asString(value: unknown): string | undefined {
if (typeof value !== "string") return undefined;
const trimmed = value.trim();
return trimmed.length > 0 ? trimmed : undefined;
}
function normalizeList(value: unknown): string[] {
if (Array.isArray(value)) {
return value
.filter((item): item is string => typeof item === "string")
.map((item) => item.trim())
.filter((item) => item.length > 0);
}
if (typeof value === "string") {
return value
.split(",")
.map((item) => item.trim())
.filter((item) => item.length > 0);
}
return [];
}
function formatLabel(value: string): string {
const cleaned = value.replace(/[-_]+/g, " ").replace(/\s+/g, " ").trim();
if (!cleaned) return value;
return cleaned.charAt(0).toUpperCase() + cleaned.slice(1);
}
function formatSection(title: string, items: string[]): string {
let md = `## ${title}\n\n`;
if (items.length === 0) {
md += "- None listed\n\n";
return md;
}
for (const item of items) {
md += `- ${formatLabel(item)}\n`;
}
md += "\n";
return md;
}
export const handleChooseALicense: SpecialHandler = async (
url: string,
timeout: number,
): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!ALLOWED_HOSTS.has(parsed.hostname)) return null;
const licenseMatch = parsed.pathname.match(LICENSE_PATH);
const isAppendix = APPENDIX_PATH.test(parsed.pathname);
if (!licenseMatch && !isAppendix) return null;
const licenseSlug = licenseMatch ? decodeURIComponent(licenseMatch[1]).toLowerCase() : "appendix";
const rawUrl = licenseMatch
? `https://raw.githubusercontent.com/github/choosealicense.com/gh-pages/_licenses/${licenseSlug}.txt`
: "https://raw.githubusercontent.com/github/choosealicense.com/gh-pages/_pages/appendix.md";
const fetchedAt = new Date().toISOString();
const result = await loadPage(rawUrl, { timeout, headers: { Accept: "text/plain" } });
if (!result.ok) return null;
const { frontmatter, body } = parseFrontmatter(result.content);
const title = asString(frontmatter.title) ?? formatLabel(licenseSlug);
const spdxId = asString(frontmatter["spdx-id"]) ?? "Unknown";
const description = asString(frontmatter.description);
const permissions = normalizeList(frontmatter.permissions);
const conditions = normalizeList(frontmatter.conditions);
const limitations = normalizeList(frontmatter.limitations);
let md = `# ${title}\n\n`;
if (description) md += `${description}\n\n`;
md += `**SPDX ID:** ${spdxId}\n`;
md += `**Source:** https://choosealicense.com${isAppendix ? "/appendix" : `/licenses/${licenseSlug}/`}\n\n`;
md += formatSection("Permissions", permissions);
md += formatSection("Conditions", conditions);
md += formatSection("Limitations", limitations);
const licenseText = body.trim();
if (licenseText.length > 0) {
md += `---\n\n## License Text\n\n${licenseText}\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "choosealicense",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Choose a License"],
};
} catch {}
return null;
};
@@ -0,0 +1,95 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
interface KevEntry {
cveID: string;
vendorProject?: string;
product?: string;
vulnerabilityName?: string;
shortDescription?: string;
requiredAction?: string;
dateAdded?: string;
dueDate?: string;
}
interface KevCatalog {
title?: string;
catalogVersion?: string;
dateReleased?: string;
count?: number;
vulnerabilities?: KevEntry[];
}
const CVE_PATTERN = /CVE-\d{4}-\d{4,7}/i;
const KEV_FEED_URL = "https://www.cisa.gov/sites/default/files/feeds/known_exploited_vulnerabilities.json";
/**
* Handle CISA Known Exploited Vulnerabilities (KEV) URLs
*/
export const handleCisaKev: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
const hostname = parsed.hostname.toLowerCase();
if (!hostname.endsWith("cisa.gov")) return null;
const path = parsed.pathname.toLowerCase();
if (!path.includes("known-exploited-vulnerabilities")) return null;
const cveMatch = parsed.pathname.match(CVE_PATTERN) ?? parsed.search.match(CVE_PATTERN);
if (!cveMatch) return null;
const cveId = cveMatch[0].toUpperCase();
const fetchedAt = new Date().toISOString();
const result = await loadPage(KEV_FEED_URL, {
timeout,
headers: { Accept: "application/json" },
});
if (!result.ok) return null;
let data: KevCatalog;
try {
data = JSON.parse(result.content) as KevCatalog;
} catch {
return null;
}
const entry = data.vulnerabilities?.find((item) => item.cveID?.toUpperCase() === cveId);
if (!entry) return null;
let md = `# ${entry.cveID}\n\n`;
if (entry.vulnerabilityName) {
md += `${entry.vulnerabilityName}\n\n`;
}
md += "## Metadata\n\n";
if (entry.vendorProject) md += `**Vendor:** ${entry.vendorProject}\n`;
if (entry.product) md += `**Product:** ${entry.product}\n`;
if (entry.dateAdded) md += `**Date Added:** ${entry.dateAdded}\n`;
if (entry.dueDate) md += `**Due Date:** ${entry.dueDate}\n`;
md += "\n";
if (entry.shortDescription) {
md += `## Description\n\n${entry.shortDescription}\n\n`;
}
if (entry.requiredAction) {
md += `## Required Action\n\n${entry.requiredAction}\n\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "cisa-kev",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via CISA KEV feed"],
};
} catch {}
return null;
};
@@ -0,0 +1,175 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, loadPage } from "./types";
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === "object" && value !== null;
}
function asString(value: unknown): string | null {
if (typeof value !== "string") return null;
const trimmed = value.trim();
return trimmed.length > 0 ? trimmed : null;
}
function asNumber(value: unknown): number | null {
return typeof value === "number" && Number.isFinite(value) ? value : null;
}
function formatLicenses(licenses: unknown): string[] {
if (!Array.isArray(licenses)) return [];
const output: string[] = [];
for (const license of licenses) {
if (typeof license === "string") {
const trimmed = license.trim();
if (trimmed) output.push(trimmed);
continue;
}
if (isRecord(license)) {
const name = asString(license.name);
const url = asString(license.url);
if (name && url) {
output.push(`${name} (${url})`);
} else if (name) {
output.push(name);
} else if (url) {
output.push(url);
}
}
}
return output;
}
function formatDependencies(deps: unknown): string[] {
const output: string[] = [];
if (Array.isArray(deps)) {
for (const dep of deps) {
if (typeof dep === "string") {
const trimmed = dep.trim();
if (trimmed) output.push(trimmed);
continue;
}
if (Array.isArray(dep)) {
const name = asString(dep[0]);
const version = asString(dep[1]);
if (name && version) {
output.push(`${name}: ${version}`);
} else if (name) {
output.push(name);
}
continue;
}
if (isRecord(dep)) {
const name = asString(dep.name) ?? asString(dep.artifact) ?? asString(dep.jar_name);
const version = asString(dep.version);
if (name && version) {
output.push(`${name}: ${version}`);
} else if (name) {
output.push(name);
}
}
}
return output;
}
if (isRecord(deps)) {
for (const [name, version] of Object.entries(deps)) {
const versionText = asString(version);
if (versionText) {
output.push(`${name}: ${versionText}`);
} else if (name.trim()) {
output.push(name);
}
}
}
return output;
}
/**
* Handle Clojars URLs via API
*/
export const handleClojars: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "clojars.org" && parsed.hostname !== "www.clojars.org") return null;
const path = parsed.pathname.replace(/^\/+|\/+$/g, "");
if (!path) return null;
const segments = path.split("/").filter(Boolean);
if (segments.length < 1 || segments.length > 2) return null;
const groupFromUrl = segments.length === 2 ? decodeURIComponent(segments[0]) : null;
const artifactFromUrl = decodeURIComponent(segments[segments.length - 1]);
const apiUrl =
segments.length === 2
? `https://clojars.org/api/artifacts/${encodeURIComponent(groupFromUrl ?? "")}/${encodeURIComponent(artifactFromUrl)}`
: `https://clojars.org/api/artifacts/${encodeURIComponent(artifactFromUrl)}`;
const fetchedAt = new Date().toISOString();
const result = await loadPage(apiUrl, {
timeout,
headers: { Accept: "application/json" },
});
if (!result.ok) return null;
let payload: unknown;
try {
payload = JSON.parse(result.content);
} catch {
return null;
}
const data = Array.isArray(payload) ? payload[0] : payload;
if (!isRecord(data)) return null;
const groupName = asString(data.group_name) ?? asString(data.group) ?? groupFromUrl;
const artifactName = asString(data.jar_name) ?? asString(data.artifact) ?? asString(data.name) ?? artifactFromUrl;
const version = asString(data.latest_version) ?? asString(data.version);
const description = asString(data.description) ?? asString(data.summary);
const downloads =
asNumber(data.downloads) ?? asNumber(data.downloads_total) ?? asNumber(data.total_downloads) ?? null;
const homepage = asString(data.homepage) ?? asString(data.url);
const licenses = formatLicenses(data.licenses);
const dependencies = formatDependencies(data.dependencies ?? data.deps);
const displayName =
groupName && artifactName && groupName !== artifactName
? `${groupName}/${artifactName}`
: (artifactName ?? groupName ?? "Clojars artifact");
let md = `# ${displayName}\n\n`;
if (description) md += `${description}\n\n`;
if (groupName) md += `**Group:** ${groupName}\n`;
if (artifactName) md += `**Artifact:** ${artifactName}\n`;
if (version) md += `**Latest:** ${version}\n`;
if (downloads !== null) md += `**Downloads:** ${formatCount(downloads)}\n`;
if (homepage) md += `**Homepage:** ${homepage}\n`;
if (licenses.length > 0) md += `**Licenses:** ${licenses.join(", ")}\n`;
if (dependencies.length > 0) {
md += "\n## Dependencies\n\n";
for (const dep of dependencies) {
md += `- ${dep}\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "clojars",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Clojars API"],
};
} catch {}
return null;
};
@@ -0,0 +1,144 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, htmlToBasicMarkdown, loadPage } from "./types";
interface CrossrefAuthor {
given?: string;
family?: string;
name?: string;
}
interface CrossrefDate {
"date-parts"?: number[][];
}
interface CrossrefMessage {
title?: string[];
author?: CrossrefAuthor[];
"container-title"?: string[];
"short-container-title"?: string[];
publisher?: string;
published?: CrossrefDate;
"published-print"?: CrossrefDate;
"published-online"?: CrossrefDate;
issued?: CrossrefDate;
created?: CrossrefDate;
DOI?: string;
abstract?: string;
type?: string;
}
interface CrossrefResponse {
message?: CrossrefMessage;
}
const DOI_HOSTS = new Set(["doi.org", "dx.doi.org", "www.doi.org"]);
function extractDoi(pathname: string): string | null {
const raw = pathname.replace(/^\/+/, "");
if (!raw) return null;
return decodeURIComponent(raw);
}
function formatAuthors(authors?: CrossrefAuthor[]): string | null {
if (!authors || authors.length === 0) return null;
const names = authors
.map((author) => {
if (author.name) return author.name;
const parts = [author.given, author.family].filter(Boolean);
return parts.length > 0 ? parts.join(" ") : null;
})
.filter((name): name is string => Boolean(name));
if (names.length === 0) return null;
return names.join(", ");
}
function formatDate(date?: CrossrefDate): string | null {
const parts = date?.["date-parts"]?.[0];
if (!parts || parts.length === 0) return null;
const [year, month, day] = parts;
if (!year) return null;
const formatted = [
String(year),
month ? String(month).padStart(2, "0") : "",
day ? String(day).padStart(2, "0") : "",
].filter(Boolean);
return formatted.join("-");
}
function formatAbstract(abstract?: string): string | null {
if (!abstract) return null;
const normalized = abstract.replace(/<\/?jats:p[^>]*>/g, (match) => (match.startsWith("</") ? "</p>" : "<p>"));
const markdown = htmlToBasicMarkdown(normalized);
return markdown.trim().length > 0 ? markdown : null;
}
export const handleCrossref: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!DOI_HOSTS.has(parsed.hostname.toLowerCase())) return null;
const doi = extractDoi(parsed.pathname);
if (!doi) return null;
const fetchedAt = new Date().toISOString();
const apiUrl = `https://api.crossref.org/works/${encodeURIComponent(doi)}`;
const result = await loadPage(apiUrl, {
timeout,
headers: {
Accept: "application/json",
},
});
if (!result.ok) return null;
let data: CrossrefResponse;
try {
data = JSON.parse(result.content);
} catch {
return null;
}
const message = data.message;
if (!message) return null;
const title = message.title?.[0]?.trim() || "CrossRef Record";
const authors = formatAuthors(message.author);
const journal = message["container-title"]?.[0] || message["short-container-title"]?.[0];
const publisher = message.publisher;
const published =
formatDate(message.published) ||
formatDate(message["published-print"]) ||
formatDate(message["published-online"]) ||
formatDate(message.issued) ||
formatDate(message.created);
const doiValue = message.DOI || doi;
const abstract = formatAbstract(message.abstract);
const type = message.type?.replace(/-/g, " ");
let md = `# ${title}\n\n`;
if (authors) md += `**Authors:** ${authors}\n`;
if (journal) md += `**Journal:** ${journal}\n`;
if (publisher) md += `**Publisher:** ${publisher}\n`;
if (published) md += `**Published:** ${published}\n`;
md += `**DOI:** ${doiValue}\n`;
if (type) md += `**Type:** ${type}\n`;
md += "\n---\n\n";
md += "## Abstract\n\n";
md += abstract || "No abstract available.";
md += "\n";
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "crossref",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via CrossRef API"],
};
} catch {}
return null;
};
@@ -0,0 +1,217 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, htmlToBasicMarkdown, loadPage } from "./types";
interface DiscourseUser {
username?: string;
name?: string;
}
interface DiscoursePost {
id: number;
username?: string;
name?: string;
created_at?: string;
cooked?: string;
raw?: string;
like_count?: number;
post_number?: number;
}
interface DiscoursePostResponse extends DiscoursePost {
topic_id?: number;
}
interface DiscourseTopic {
id?: number;
title?: string;
fancy_title?: string;
posts_count?: number;
created_at?: string;
views?: number;
like_count?: number;
tags?: string[];
category_id?: number;
category_slug?: string;
category?: { id?: number; name?: string; slug?: string };
excerpt?: string;
details?: { created_by?: DiscourseUser };
post_stream?: { posts?: DiscoursePost[] };
}
const MAX_POSTS = 20;
function normalizeBasePath(basePath: string): string {
if (!basePath || basePath === "/") return "";
return basePath.replace(/\/$/, "");
}
function parseTopicPath(pathname: string): { basePath: string; topicId: string } | null {
const match = pathname.match(/^(.*?)(?:\/t\/)(?:[^/]+\/)?(\d+)(?:\.json)?(?:\/|$)/);
if (!match) return null;
return { basePath: match[1] ?? "", topicId: match[2] };
}
function parsePostPath(pathname: string): { basePath: string; postId: string } | null {
const match = pathname.match(/^(.*?)(?:\/posts\/)(\d+)(?:\.json)?(?:\/|$)/);
if (!match) return null;
return { basePath: match[1] ?? "", postId: match[2] };
}
function formatAuthor(user?: DiscourseUser | null): string {
if (!user) return "unknown";
const name = user.name?.trim();
const username = user.username?.trim();
if (name && username && name !== username) return `${name} (@${username})`;
if (username) return `@${username}`;
if (name) return name;
return "unknown";
}
function formatIsoDate(value?: string): string {
if (!value) return "unknown";
const date = new Date(value);
if (Number.isNaN(date.getTime())) return value;
return date.toISOString().split("T")[0];
}
function formatCategory(topic: DiscourseTopic): string | null {
const parts: string[] = [];
const name = topic.category?.name ?? topic.category_slug;
if (name) parts.push(name);
const id = topic.category?.id ?? topic.category_id;
if (id != null) parts.push(`#${id}`);
return parts.length ? parts.join(" ") : null;
}
function formatPostBody(post: DiscoursePost): string {
const raw = post.raw?.trim();
if (raw) return raw;
const cooked = post.cooked?.trim();
if (!cooked) return "";
return htmlToBasicMarkdown(cooked);
}
function buildTopicUrl(baseUrl: string, topicId: string): string {
const topicUrl = new URL(`${baseUrl}/t/${topicId}.json`);
topicUrl.searchParams.set("include_raw", "1");
return topicUrl.toString();
}
function buildPostUrl(baseUrl: string, postId: string): string {
const postUrl = new URL(`${baseUrl}/posts/${postId}.json`);
postUrl.searchParams.set("include_raw", "1");
return postUrl.toString();
}
/**
* Handle Discourse forum URLs via API
*/
export const handleDiscourse: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
const topicMatch = parseTopicPath(parsed.pathname);
const postMatch = topicMatch ? null : parsePostPath(parsed.pathname);
if (!topicMatch && !postMatch) return null;
const basePath = normalizeBasePath(topicMatch?.basePath ?? postMatch?.basePath ?? "");
const baseUrl = `${parsed.origin}${basePath}`;
let requestedPost: DiscoursePost | null = null;
let topicId = topicMatch?.topicId ?? null;
if (!topicId && postMatch) {
const postResult = await loadPage(buildPostUrl(baseUrl, postMatch.postId), { timeout });
if (!postResult.ok) return null;
let postData: DiscoursePostResponse;
try {
postData = JSON.parse(postResult.content) as DiscoursePostResponse;
} catch {
return null;
}
if (!postData.topic_id) return null;
topicId = String(postData.topic_id);
requestedPost = postData;
}
if (!topicId) return null;
const topicResult = await loadPage(buildTopicUrl(baseUrl, topicId), { timeout });
if (!topicResult.ok) return null;
let topic: DiscourseTopic;
try {
topic = JSON.parse(topicResult.content) as DiscourseTopic;
} catch {
return null;
}
const title = topic.title || topic.fancy_title;
if (!title) return null;
const fetchedAt = new Date().toISOString();
const posts: DiscoursePost[] = [...(topic.post_stream?.posts ?? [])];
if (requestedPost && !posts.some((post) => post.id === requestedPost?.id)) {
posts.unshift(requestedPost);
}
let md = `# ${title}\n\n`;
const metaParts: string[] = [];
if (topic.id != null) metaParts.push(`**Topic ID:** ${topic.id}`);
if (topic.posts_count != null) metaParts.push(`**Posts:** ${topic.posts_count}`);
if (topic.views != null) metaParts.push(`**Views:** ${topic.views}`);
if (topic.like_count != null) metaParts.push(`**Likes:** ${topic.like_count}`);
if (metaParts.length) md += `${metaParts.join(" | ")}\n`;
const categoryLabel = formatCategory(topic);
if (categoryLabel) md += `**Category:** ${categoryLabel}\n`;
if (topic.tags?.length) md += `**Tags:** ${topic.tags.join(", ")}\n`;
const createdBy = formatAuthor(topic.details?.created_by ?? null);
if (createdBy !== "unknown" || topic.created_at) {
md += `**Created by:** ${createdBy} - ${formatIsoDate(topic.created_at)}\n`;
}
md += "\n";
const description = topic.excerpt
? htmlToBasicMarkdown(topic.excerpt)
: posts.length
? formatPostBody(posts[0])
: "";
if (description) {
md += `## Description\n\n${description}\n\n`;
}
if (posts.length) {
md += "## Posts\n\n";
for (const post of posts.slice(0, MAX_POSTS)) {
const author = formatAuthor({ name: post.name, username: post.username });
const date = formatIsoDate(post.created_at);
const likes = post.like_count ?? 0;
const content = formatPostBody(post);
const postLabel = post.post_number != null ? `Post ${post.post_number}` : `Post ${post.id}`;
md += `### ${postLabel} - ${author} - ${date} - Likes: ${likes}\n\n`;
md += content ? `${content}\n\n---\n\n` : "_No content available._\n\n---\n\n";
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "discourse-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Discourse API"],
};
} catch {}
return null;
};
@@ -0,0 +1,153 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
type LocalizedText = string | Record<string, string>;
type FdroidPackage = {
packageName?: string;
name?: LocalizedText;
summary?: LocalizedText;
description?: LocalizedText;
author?: string | { name?: string; email?: string };
authorName?: string;
authorEmail?: string;
license?: string;
categories?: string[];
antiFeatures?: string[];
sourceCode?: string;
packages?: Array<{
versionName?: string;
versionCode?: number;
added?: number;
antiFeatures?: string[];
}>;
suggestedVersionCode?: number;
suggestedVersionName?: string;
};
function pickLocalizedText(value?: LocalizedText): string | undefined {
if (!value) return undefined;
if (typeof value === "string") return value;
const preferred = value["en-US"] ?? value.en_US ?? value.en;
if (preferred) return preferred;
const first = Object.values(value).find((entry) => typeof entry === "string");
return first;
}
function normalizeAuthor(data: FdroidPackage): string | undefined {
if (data.authorName) return data.authorName;
if (typeof data.author === "string") return data.author;
if (data.author && typeof data.author !== "string" && typeof data.author.name === "string") return data.author.name;
if (data.authorEmail) return data.authorEmail;
return undefined;
}
function normalizeAuthorEmail(data: FdroidPackage): string | undefined {
if (data.authorEmail) return data.authorEmail;
if (data.author && typeof data.author !== "string" && typeof data.author.email === "string")
return data.author.email;
return undefined;
}
function collectAntiFeatures(data: FdroidPackage): string[] {
const values = new Set<string>();
for (const feature of data.antiFeatures ?? []) values.add(feature);
for (const pkg of data.packages ?? []) {
for (const feature of pkg.antiFeatures ?? []) values.add(feature);
}
return Array.from(values);
}
function resolveSuggestedVersion(data: FdroidPackage): string | undefined {
if (data.suggestedVersionName) return data.suggestedVersionName;
if (data.suggestedVersionCode) {
const match = data.packages?.find((pkg) => pkg.versionCode === data.suggestedVersionCode);
if (match?.versionName) return match.versionName;
}
return data.packages?.[0]?.versionName;
}
/**
* Handle F-Droid URLs via API
*/
export const handleFdroid: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "f-droid.org" && parsed.hostname !== "www.f-droid.org") return null;
// Extract package name from /packages/{packageName} or /en/packages/{packageName}
const match = parsed.pathname.match(/^\/(?:en\/)?packages\/([^/]+)/);
if (!match) return null;
const packageName = decodeURIComponent(match[1]);
const fetchedAt = new Date().toISOString();
const apiUrl = `https://f-droid.org/api/v1/packages/${encodeURIComponent(packageName)}`;
const result = await loadPage(apiUrl, {
timeout,
headers: { Accept: "application/json" },
});
if (!result.ok) return null;
let data: FdroidPackage;
try {
data = JSON.parse(result.content) as FdroidPackage;
} catch {
return null;
}
const displayName = pickLocalizedText(data.name) ?? packageName;
const summary = pickLocalizedText(data.summary);
const description = pickLocalizedText(data.description);
const author = normalizeAuthor(data);
const authorEmail = normalizeAuthorEmail(data);
const antiFeatures = collectAntiFeatures(data);
const latestVersion = resolveSuggestedVersion(data);
let md = `# ${displayName}\n\n`;
if (summary) md += `${summary}\n\n`;
md += `**Package:** ${packageName}`;
if (latestVersion) md += ` · **Latest:** ${latestVersion}`;
if (data.license) md += ` · **License:** ${data.license}`;
md += "\n";
if (author) {
md += `**Author:** ${author}`;
if (authorEmail && authorEmail !== author) md += ` <${authorEmail}>`;
md += "\n";
}
if (data.sourceCode) md += `**Source Code:** ${data.sourceCode}\n`;
if (data.categories?.length) md += `**Categories:** ${data.categories.join(", ")}\n`;
if (antiFeatures.length) md += `**Anti-Features:** ${antiFeatures.join(", ")}\n`;
if (description) {
md += `\n## Description\n\n${description}\n`;
}
if (data.packages?.length) {
md += "\n## Version History\n\n";
for (const version of data.packages.slice(0, 10)) {
const label = version.versionName ?? "unknown";
const code = version.versionCode ? ` (${version.versionCode})` : "";
md += `- ${label}${code}\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "fdroid",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via F-Droid API"],
};
} catch {}
return null;
};
@@ -0,0 +1,213 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, htmlToBasicMarkdown, loadPage } from "./types";
type LocalizedText = string | Record<string, string | null | undefined> | null | undefined;
type AddonFile = {
permissions?: string[];
host_permissions?: string[];
optional_permissions?: string[];
optional_host_permissions?: string[];
};
type AddonLicense = {
name?: LocalizedText;
slug?: string;
url?: string;
};
type AddonVersion = {
version?: string;
license?: AddonLicense;
file?: AddonFile;
};
type AddonHomepage = {
url?: LocalizedText;
outgoing?: LocalizedText;
};
type AddonData = {
name?: LocalizedText;
summary?: LocalizedText;
description?: LocalizedText;
default_locale?: string;
authors?: Array<{ name?: string | null }>;
average_daily_users?: number;
weekly_downloads?: number;
ratings?: { average?: number; count?: number };
current_version?: AddonVersion;
categories?: string[] | Record<string, string[]>;
homepage?: AddonHomepage;
url?: string;
};
function getLocalizedText(value: LocalizedText, defaultLocale?: string): string | undefined {
if (!value) return undefined;
if (typeof value === "string") return value;
const localized = value as Record<string, string | null | undefined>;
if (defaultLocale && localized[defaultLocale]) return localized[defaultLocale] ?? undefined;
if (localized["en-US"]) return localized["en-US"] ?? undefined;
for (const entry of Object.values(localized)) {
if (entry) return entry;
}
return undefined;
}
function normalizeCategories(categories?: string[] | Record<string, string[]>): string[] {
if (!categories) return [];
if (Array.isArray(categories)) return categories.filter(Boolean);
const values: string[] = [];
for (const list of Object.values(categories)) {
if (Array.isArray(list)) {
for (const item of list) {
if (item) values.push(item);
}
}
}
const seen = new Set<string>();
return values.filter((item) => {
if (seen.has(item)) return false;
seen.add(item);
return true;
});
}
function collectPermissions(file?: AddonFile): string[] {
if (!file) return [];
const permissions: string[] = [];
const seen = new Set<string>();
const add = (items?: string[]) => {
for (const item of items ?? []) {
if (!item || seen.has(item)) continue;
seen.add(item);
permissions.push(item);
}
};
add(file.permissions);
add(file.host_permissions);
add(file.optional_permissions);
add(file.optional_host_permissions);
return permissions;
}
export const handleFirefoxAddons: SpecialHandler = async (
url: string,
timeout: number,
): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "addons.mozilla.org") return null;
const segments = parsed.pathname.split("/").filter(Boolean);
const addonIndex = segments.indexOf("addon");
if (addonIndex === -1) return null;
const slug = segments[addonIndex + 1] ? decodeURIComponent(segments[addonIndex + 1]) : "";
if (!slug) return null;
const apiUrl = `https://addons.mozilla.org/api/v5/addons/addon/${encodeURIComponent(slug)}/`;
const result = await loadPage(apiUrl, { timeout, headers: { Accept: "application/json" } });
if (!result.ok) return null;
let data: AddonData;
try {
data = JSON.parse(result.content) as AddonData;
} catch {
return null;
}
const fetchedAt = new Date().toISOString();
const defaultLocale = data.default_locale || "en-US";
const name = getLocalizedText(data.name, defaultLocale) ?? slug;
const summary = getLocalizedText(data.summary, defaultLocale);
const descriptionRaw = getLocalizedText(data.description, defaultLocale);
const description = descriptionRaw ? htmlToBasicMarkdown(descriptionRaw) : undefined;
const authors = (data.authors ?? [])
.map((author) => author.name ?? "")
.map((author) => author.trim())
.filter(Boolean);
const ratingAverage = data.ratings?.average;
const ratingCount = data.ratings?.count;
const users = data.average_daily_users ?? data.weekly_downloads;
const version = data.current_version?.version;
const categories = normalizeCategories(data.categories);
const licenseName =
getLocalizedText(data.current_version?.license?.name, defaultLocale) ?? data.current_version?.license?.slug;
const licenseUrl = data.current_version?.license?.url;
const homepage =
getLocalizedText(data.homepage?.url, defaultLocale) ??
getLocalizedText(data.homepage?.outgoing, defaultLocale);
const permissions = collectPermissions(data.current_version?.file);
let md = `# ${name}\n\n`;
if (summary) md += `${summary}\n\n`;
if (authors.length > 0) {
md += `**Author${authors.length > 1 ? "s" : ""}:** ${authors.join(", ")}\n`;
}
if (ratingAverage !== undefined) {
md += `**Rating:** ${ratingAverage.toFixed(2)}`;
if (ratingCount !== undefined) md += ` (${formatCount(ratingCount)} reviews)`;
md += "\n";
}
if (users !== undefined) md += `**Users:** ${formatCount(users)}\n`;
if (version) md += `**Version:** ${version}\n`;
if (categories.length > 0) md += `**Categories:** ${categories.join(", ")}\n`;
if (licenseName && licenseUrl) {
md += `**License:** [${licenseName}](${licenseUrl})\n`;
} else if (licenseName) {
md += `**License:** ${licenseName}\n`;
} else if (licenseUrl) {
md += `**License:** ${licenseUrl}\n`;
}
if (homepage) md += `**Homepage:** ${homepage}\n`;
if (description) {
md += `\n## Description\n\n${description}\n`;
}
if (permissions.length > 0) {
const preview = permissions.slice(0, 40);
md += `\n## Permissions (${permissions.length})\n\n`;
for (const permission of preview) {
md += `- ${permission}\n`;
}
if (permissions.length > preview.length) {
md += `\n*...and ${permissions.length - preview.length} more*\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: data.url ?? result.finalUrl ?? url,
contentType: "text/markdown",
method: "firefox-addons",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Firefox Add-ons API"],
};
} catch {}
return null;
};
@@ -0,0 +1,235 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, htmlToBasicMarkdown, loadPage } from "./types";
interface FlathubScreenshotSize {
src?: string;
width?: string;
height?: string;
scale?: string;
}
interface FlathubScreenshot {
caption?: string | null;
sizes?: FlathubScreenshotSize[];
}
interface FlathubRelease {
version?: string;
timestamp?: string;
description?: string | null;
url?: string | null;
type?: string | null;
}
interface FlathubAppStream {
id?: string;
name?: string;
summary?: string;
description?: string;
developer_name?: string;
categories?: string[];
screenshots?: FlathubScreenshot[];
releases?: FlathubRelease[];
metadata?: Record<string, unknown>;
installs?: number | string;
permissions?: unknown;
}
function extractAppId(pathname: string): string | null {
const detailsMatch = pathname.match(/^\/apps\/details\/([^/]+)\/?$/);
if (detailsMatch) return decodeURIComponent(detailsMatch[1]);
const appMatch = pathname.match(/^\/apps\/([^/]+)\/?$/);
if (appMatch) return decodeURIComponent(appMatch[1]);
return null;
}
function parseNumber(value: unknown): number | null {
if (typeof value === "number" && Number.isFinite(value)) return value;
if (typeof value === "string") {
const cleaned = value.replace(/[^0-9.]/g, "");
if (!cleaned) return null;
const parsed = Number(cleaned);
if (!Number.isNaN(parsed)) return parsed;
}
return null;
}
function normalizeStringList(value: unknown): string[] {
if (Array.isArray(value)) {
return value.filter((item): item is string => typeof item === "string" && item.trim().length > 0);
}
if (typeof value === "string") {
return value
.split(/[,;\n]+/)
.map((item) => item.trim())
.filter(Boolean);
}
return [];
}
function extractInstalls(app: FlathubAppStream): number | null {
const direct = parseNumber(app.installs);
if (direct !== null) return direct;
if (!app.metadata) return null;
for (const [key, value] of Object.entries(app.metadata)) {
if (!key.toLowerCase().includes("install")) continue;
const parsed = parseNumber(value);
if (parsed !== null) return parsed;
}
return null;
}
function extractPermissions(app: FlathubAppStream): string[] {
const permissions: string[] = [];
permissions.push(...normalizeStringList(app.permissions));
if (app.metadata) {
for (const [key, value] of Object.entries(app.metadata)) {
if (!key.toLowerCase().includes("permission")) continue;
const list = normalizeStringList(value);
if (list.length) {
permissions.push(...list);
continue;
}
if (typeof value === "string" || typeof value === "number" || typeof value === "boolean") {
permissions.push(`${key}: ${String(value)}`);
}
}
}
return Array.from(new Set(permissions));
}
function screenshotArea(size?: FlathubScreenshotSize): number {
if (!size) return 0;
const width = Number(size.width);
const height = Number(size.height);
if (!Number.isFinite(width) || !Number.isFinite(height)) return 0;
return width * height;
}
function bestScreenshotUrl(sizes?: FlathubScreenshotSize[]): string | null {
if (!sizes || sizes.length === 0) return null;
let best = sizes[0];
let bestArea = screenshotArea(best);
for (const size of sizes) {
const area = screenshotArea(size);
if (area > bestArea) {
best = size;
bestArea = area;
}
}
return best.src ?? sizes[0].src ?? null;
}
function formatReleaseDate(timestamp?: string | null): string | null {
if (!timestamp) return null;
const seconds = Number(timestamp);
if (!Number.isFinite(seconds)) return null;
const date = new Date(seconds * 1000);
if (Number.isNaN(date.getTime())) return null;
return date.toISOString().split("T")[0] ?? null;
}
export const handleFlathub: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "flathub.org" && parsed.hostname !== "www.flathub.org") return null;
const appId = extractAppId(parsed.pathname);
if (!appId) return null;
const apiUrl = `https://flathub.org/api/v2/appstream/${encodeURIComponent(appId)}`;
const result = await loadPage(apiUrl, { timeout, headers: { Accept: "application/json" } });
if (!result.ok) return null;
let app: FlathubAppStream;
try {
app = JSON.parse(result.content) as FlathubAppStream;
} catch {
return null;
}
const fetchedAt = new Date().toISOString();
const name = app.name ?? app.id ?? appId;
let md = `# ${name}\n\n`;
if (app.summary) md += `${app.summary}\n\n`;
md += "## Metadata\n\n";
md += `**App ID:** ${app.id ?? appId}\n`;
if (app.developer_name) md += `**Developer:** ${app.developer_name}\n`;
const installs = extractInstalls(app);
if (installs !== null) md += `**Installs:** ${formatCount(installs)}\n`;
if (app.categories?.length) {
md += "\n## Categories\n\n";
for (const category of app.categories) {
md += `- ${category}\n`;
}
}
if (app.description) {
const description = htmlToBasicMarkdown(app.description);
if (description) md += `\n## Description\n\n${description}\n`;
}
const permissions = extractPermissions(app);
if (permissions.length) {
md += "\n## Permissions\n\n";
for (const permission of permissions) {
md += `- ${permission}\n`;
}
}
if (app.screenshots?.length) {
md += "\n## Screenshots\n\n";
for (const screenshot of app.screenshots.slice(0, 5)) {
const screenshotUrl = bestScreenshotUrl(screenshot.sizes);
if (!screenshotUrl) continue;
const caption = screenshot.caption ? ` - ${screenshot.caption}` : "";
md += `- ${screenshotUrl}${caption}\n`;
}
}
if (app.releases?.length) {
md += "\n## Releases\n\n";
for (const release of app.releases.slice(0, 5)) {
const version = release.version ?? "unknown";
let line = `- **${version}**`;
const date = formatReleaseDate(release.timestamp);
if (date) line += ` (${date})`;
if (release.type) line += ` · ${release.type}`;
if (release.url) line += ` · ${release.url}`;
md += `${line}\n`;
if (release.description) {
const releaseDesc = htmlToBasicMarkdown(release.description).replace(/\n+/g, " ").trim();
if (releaseDesc) md += ` - ${releaseDesc}\n`;
}
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: result.finalUrl,
contentType: "text/markdown",
method: "flathub-appstream",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Flathub Appstream API"],
};
} catch {}
return null;
};
@@ -64,7 +64,7 @@ function parseGitHubUrl(url: string): GitHubUrl | null {
* Convert GitHub blob URL to raw URL
*/
function toRawGitHubUrl(gh: GitHubUrl): string {
return `https://raw.githubusercontent.com/${gh.owner}/${gh.repo}/refs/heads/${gh.ref}/${gh.path}`;
return `https://raw.githubusercontent.com/${gh.owner}/${gh.repo}/${gh.ref}/${gh.path}`;
}
/**
@@ -72,8 +72,7 @@ function toRawGitHubUrl(gh: GitHubUrl): string {
*/
export async function fetchGitHubApi(endpoint: string, timeout: number): Promise<{ data: unknown; ok: boolean }> {
try {
const controller = new AbortController();
const timeoutId = setTimeout(() => controller.abort(), timeout * 1000);
const timeoutSignal = AbortSignal.timeout(timeout * 1000);
const headers: Record<string, string> = {
Accept: "application/vnd.github.v3+json",
@@ -87,12 +86,10 @@ export async function fetchGitHubApi(endpoint: string, timeout: number): Promise
}
const response = await fetch(`https://api.github.com${endpoint}`, {
signal: controller.signal,
signal: timeoutSignal,
headers,
});
clearTimeout(timeoutId);
if (!response.ok) {
return { data: null, ok: false };
}
@@ -240,7 +237,7 @@ async function renderGitHubTree(gh: GitHubUrl, timeout: number): Promise<{ conte
const readmeFile = items.find((item) => item.type === "file" && /^readme\.md$/i.test(item.name));
if (readmeFile) {
const readmePath = dirPath ? `${dirPath}/${readmeFile.name}` : readmeFile.name;
const rawUrl = `https://raw.githubusercontent.com/${gh.owner}/${gh.repo}/refs/heads/${ref}/${readmePath}`;
const rawUrl = `https://raw.githubusercontent.com/${gh.owner}/${gh.repo}/${ref}/${readmePath}`;
const readmeResult = await loadPage(rawUrl, { timeout });
if (readmeResult.ok) {
md += `---\n\n## README\n\n${readmeResult.content}`;
@@ -0,0 +1,250 @@
/**
* Web Fetch Special Handlers Index
*
* Exports all special handlers for site-specific content extraction.
*/
import { handleArtifactHub } from "./artifacthub";
import { handleArxiv } from "./arxiv";
import { handleAur } from "./aur";
import { handleBiorxiv } from "./biorxiv";
import { handleBluesky } from "./bluesky";
import { handleBrew } from "./brew";
import { handleCheatSh } from "./cheatsh";
import { handleChocolatey } from "./chocolatey";
import { handleChooseALicense } from "./choosealicense";
import { handleCisaKev } from "./cisa-kev";
import { handleClojars } from "./clojars";
import { handleCoinGecko } from "./coingecko";
import { handleCratesIo } from "./crates-io";
import { handleCrossref } from "./crossref";
import { handleDevTo } from "./devto";
import { handleDiscogs } from "./discogs";
import { handleDiscourse } from "./discourse";
import { handleDockerHub } from "./dockerhub";
import { handleFdroid } from "./fdroid";
import { handleFirefoxAddons } from "./firefox-addons";
import { handleFlathub } from "./flathub";
import { fetchGitHubApi, handleGitHub } from "./github";
import { handleGitHubGist } from "./github-gist";
import { handleGitLab } from "./gitlab";
import { handleGoPkg } from "./go-pkg";
import { handleHackage } from "./hackage";
import { handleHackerNews } from "./hackernews";
import { handleHex } from "./hex";
import { handleHuggingFace } from "./huggingface";
import { handleIacr } from "./iacr";
import { handleJetBrainsMarketplace } from "./jetbrains-marketplace";
import { handleLemmy } from "./lemmy";
import { handleLobsters } from "./lobsters";
import { handleMastodon } from "./mastodon";
import { handleMaven } from "./maven";
import { handleMDN } from "./mdn";
import { handleMetaCPAN } from "./metacpan";
import { handleMusicBrainz } from "./musicbrainz";
import { handleNpm } from "./npm";
import { handleNuGet } from "./nuget";
import { handleNvd } from "./nvd";
import { handleOllama } from "./ollama";
import { handleOpenVsx } from "./open-vsx";
import { handleOpenCorporates } from "./opencorporates";
import { handleOpenLibrary } from "./openlibrary";
import { handleOrcid } from "./orcid";
import { handleOsv } from "./osv";
import { handlePackagist } from "./packagist";
import { handlePubDev } from "./pub-dev";
import { handlePubMed } from "./pubmed";
import { handlePyPI } from "./pypi";
import { handleRawg } from "./rawg";
import { handleReadTheDocs } from "./readthedocs";
import { handleReddit } from "./reddit";
import { handleRepology } from "./repology";
import { handleRfc } from "./rfc";
import { handleRubyGems } from "./rubygems";
import { handleSearchcode } from "./searchcode";
import { handleSecEdgar } from "./sec-edgar";
import { handleSemanticScholar } from "./semantic-scholar";
import { handleSnapcraft } from "./snapcraft";
import { handleSourcegraph } from "./sourcegraph";
import { handleSpdx } from "./spdx";
import { handleSpotify } from "./spotify";
import { handleStackOverflow } from "./stackoverflow";
import { handleTerraform } from "./terraform";
import { handleTldr } from "./tldr";
import { handleTwitter } from "./twitter";
import type { SpecialHandler } from "./types";
import { handleVimeo } from "./vimeo";
import { handleVscodeMarketplace } from "./vscode-marketplace";
import { handleW3c } from "./w3c";
import { handleWikidata } from "./wikidata";
import { handleWikipedia } from "./wikipedia";
import { handleYouTube } from "./youtube";
export type { RenderResult, SpecialHandler } from "./types";
export {
fetchGitHubApi,
handleArtifactHub,
handleArxiv,
handleAur,
handleBiorxiv,
handleBluesky,
handleBrew,
handleCheatSh,
handleCisaKev,
handleChocolatey,
handleClojars,
handleChooseALicense,
handleCoinGecko,
handleCratesIo,
handleCrossref,
handleDevTo,
handleDiscogs,
handleDiscourse,
handleDockerHub,
handleFdroid,
handleFlathub,
handleFirefoxAddons,
handleGitHub,
handleGitHubGist,
handleGitLab,
handleGoPkg,
handleHackage,
handleHackerNews,
handleHex,
handleHuggingFace,
handleIacr,
handleJetBrainsMarketplace,
handleLemmy,
handleLobsters,
handleMastodon,
handleMaven,
handleMDN,
handleMetaCPAN,
handleMusicBrainz,
handleNpm,
handleNuGet,
handleNvd,
handleOllama,
handleOpenCorporates,
handleOpenLibrary,
handleOrcid,
handleOpenVsx,
handleOsv,
handlePackagist,
handlePubDev,
handlePubMed,
handlePyPI,
handleRawg,
handleReadTheDocs,
handleReddit,
handleRepology,
handleRfc,
handleRubyGems,
handleSecEdgar,
handleSearchcode,
handleSemanticScholar,
handleSnapcraft,
handleSourcegraph,
handleSpotify,
handleSpdx,
handleStackOverflow,
handleTerraform,
handleTldr,
handleTwitter,
handleVimeo,
handleVscodeMarketplace,
handleW3c,
handleWikidata,
handleWikipedia,
handleYouTube,
};
export const specialHandlers: SpecialHandler[] = [
// Git hosting
handleGitHubGist,
handleGitHub,
handleGitLab,
// Video/Media
handleYouTube,
handleVimeo,
handleSpotify,
handleDiscogs,
handleMusicBrainz,
// Games
handleRawg,
// Social/News
handleTwitter,
handleBluesky,
handleMastodon,
handleLemmy,
handleHackerNews,
handleLobsters,
handleReddit,
handleDiscourse,
// Developer content
handleStackOverflow,
handleDevTo,
handleMDN,
handleReadTheDocs,
handleSearchcode,
handleSourcegraph,
handleTldr,
handleCheatSh,
// Package registries
handleNpm,
handleFirefoxAddons,
handleVscodeMarketplace,
handleNuGet,
handleChocolatey,
handleClojars,
handleBrew,
handlePyPI,
handleCratesIo,
handleDockerHub,
handleFdroid,
handleFlathub,
handleGoPkg,
handleHex,
handlePackagist,
handlePubDev,
handleMaven,
handleJetBrainsMarketplace,
handleOpenVsx,
handleArtifactHub,
handleRubyGems,
handleTerraform,
handleAur,
handleHackage,
handleMetaCPAN,
handleRepology,
handleSnapcraft,
// ML/AI
handleHuggingFace,
handleOllama,
// Academic
handleArxiv,
handleBiorxiv,
handleCrossref,
handleIacr,
handleOrcid,
handleSemanticScholar,
handlePubMed,
handleRfc,
// Security
handleCisaKev,
handleNvd,
handleOsv,
// Crypto
handleCoinGecko,
// Business
handleOpenCorporates,
handleSecEdgar,
// Reference
handleOpenLibrary,
handleChooseALicense,
handleW3c,
handleSpdx,
handleWikidata,
handleWikipedia,
];
@@ -0,0 +1,168 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, htmlToBasicMarkdown, loadPage } from "./types";
interface PluginVendor {
name?: string;
publicName?: string;
url?: string;
}
interface PluginTag {
name?: string;
}
type PluginRating =
| number
| {
rating?: number;
value?: number;
score?: number;
votes?: number;
totalVotes?: number;
count?: number;
};
interface PluginData {
id?: number;
name?: string;
description?: string;
preview?: string;
vendor?: PluginVendor;
rating?: PluginRating;
ratingCount?: number;
downloads?: number;
tags?: PluginTag[];
urls?: {
url?: string;
docUrl?: string;
sourceCodeUrl?: string;
bugtrackerUrl?: string;
};
}
interface UpdateData {
version?: string;
since?: string;
until?: string;
sinceUntil?: string;
channel?: string;
downloads?: number;
compatibleVersions?: Record<string, string>;
cdate?: string | number;
}
const MARKETPLACE_HOSTS = new Set(["plugins.jetbrains.com"]);
function extractRating(plugin: PluginData): { value: number | null; votes: number | null } {
const rating = plugin.rating;
if (typeof rating === "number" && Number.isFinite(rating)) {
return { value: rating, votes: plugin.ratingCount ?? null };
}
if (rating && typeof rating === "object") {
const value = rating.rating ?? rating.value ?? rating.score ?? null;
const votes = rating.votes ?? rating.totalVotes ?? rating.count ?? plugin.ratingCount ?? null;
return { value: typeof value === "number" ? value : null, votes: typeof votes === "number" ? votes : null };
}
return { value: null, votes: plugin.ratingCount ?? null };
}
function formatBuildCompatibility(update: UpdateData): string | null {
if (update.sinceUntil) return update.sinceUntil;
if (update.since && update.until) return `${update.since} - ${update.until}`;
if (update.since) return `${update.since}+`;
return null;
}
export const handleJetBrainsMarketplace: SpecialHandler = async (
url: string,
timeout: number,
): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!MARKETPLACE_HOSTS.has(parsed.hostname)) return null;
const match = parsed.pathname.match(/^\/plugin\/(\d+)(?:-[^/]+)?(?:\/|$)/);
if (!match) return null;
const pluginId = match[1];
const fetchedAt = new Date().toISOString();
const pluginUrl = `https://plugins.jetbrains.com/api/plugins/${pluginId}`;
const updatesUrl = `https://plugins.jetbrains.com/api/plugins/${pluginId}/updates?size=1`;
const [pluginResult, updatesResult] = await Promise.all([
loadPage(pluginUrl, { timeout }),
loadPage(updatesUrl, { timeout }),
]);
if (!pluginResult.ok || !updatesResult.ok) return null;
let plugin: PluginData;
let updates: UpdateData[];
try {
plugin = JSON.parse(pluginResult.content) as PluginData;
updates = JSON.parse(updatesResult.content) as UpdateData[];
} catch {
return null;
}
const update = updates[0];
if (!plugin?.name) return null;
const vendorName = plugin.vendor?.name ?? plugin.vendor?.publicName;
const descriptionSource = plugin.description ?? plugin.preview ?? "";
const description = descriptionSource ? htmlToBasicMarkdown(descriptionSource) : "";
const tags = (plugin.tags ?? []).map((tag) => tag.name).filter(Boolean) as string[];
const rating = extractRating(plugin);
const buildCompatibility = update ? formatBuildCompatibility(update) : null;
let md = `# ${plugin.name}\n\n`;
if (description) md += `${description}\n\n`;
md += `**Plugin ID:** ${pluginId}\n`;
if (vendorName) md += `**Vendor:** ${vendorName}\n`;
if (plugin.downloads !== undefined) {
md += `**Downloads:** ${formatCount(plugin.downloads)}\n`;
}
if (rating.value !== null) {
md += `**Rating:** ${rating.value.toFixed(2)}`;
if (rating.votes !== null) md += ` (${formatCount(rating.votes)} votes)`;
md += "\n";
}
if (tags.length) md += `**Tags:** ${tags.join(", ")}\n`;
if (update) {
md += "\n## Latest Release\n\n";
if (update.version) md += `**Version:** ${update.version}\n`;
if (update.channel) md += `**Channel:** ${update.channel}\n`;
if (buildCompatibility) md += `**Build Compatibility:** ${buildCompatibility}\n`;
if (update.downloads !== undefined) {
md += `**Release Downloads:** ${formatCount(update.downloads)}\n`;
}
}
const compatibility = update?.compatibleVersions ?? {};
const compatibilityEntries = Object.entries(compatibility).sort(([a], [b]) => a.localeCompare(b));
if (compatibilityEntries.length) {
md += "\n## IDE Compatibility\n\n";
for (const [product, version] of compatibilityEntries) {
md += `- ${product}: ${version}\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "jetbrains-marketplace",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via JetBrains Marketplace API"],
};
} catch {}
return null;
};
@@ -0,0 +1,216 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
interface LemmyCreator {
name: string;
actor_id?: string;
}
interface LemmyCommunity {
name: string;
actor_id?: string;
}
interface LemmyCounts {
score: number;
comments?: number;
}
interface LemmyPost {
id: number;
name: string;
body?: string;
url?: string;
}
interface LemmyPostView {
post: LemmyPost;
creator: LemmyCreator;
community: LemmyCommunity;
counts: LemmyCounts;
}
interface LemmyPostResponse {
post_view?: LemmyPostView;
}
interface LemmyComment {
id: number;
content?: string;
path?: string;
parent_id?: number | null;
post_id?: number;
}
interface LemmyCommentView {
comment: LemmyComment;
creator: LemmyCreator;
counts: LemmyCounts;
}
interface LemmyCommentListResponse {
comments?: LemmyCommentView[];
}
interface LemmyCommentResponse {
comment_view?: LemmyCommentView;
}
function parseJson<T>(content: string): T | null {
try {
return JSON.parse(content) as T;
} catch {
return null;
}
}
function formatCommunity(community: LemmyCommunity): string {
if (community.actor_id) {
try {
const host = new URL(community.actor_id).hostname;
return `!${community.name}@${host}`;
} catch {}
}
return `!${community.name}`;
}
function formatAuthor(creator: LemmyCreator): string {
if (creator.actor_id) {
try {
const host = new URL(creator.actor_id).hostname;
return `@${creator.name}@${host}`;
} catch {}
}
return creator.name;
}
function indentBlock(text: string, indent: string): string {
return text
.split("\n")
.map((line) => `${indent}${line}`)
.join("\n");
}
function renderComments(comments: LemmyCommentView[]): string {
const childrenByParent = new Map<number, LemmyCommentView[]>();
const commentIds = new Set(comments.map((view) => view.comment.id));
for (const commentView of comments) {
const parentId = commentView.comment.parent_id;
const resolvedParent = parentId && commentIds.has(parentId) ? parentId : 0;
const list = childrenByParent.get(resolvedParent);
if (list) {
list.push(commentView);
} else {
childrenByParent.set(resolvedParent, [commentView]);
}
}
const renderThread = (parentId: number, depth: number): string => {
const items = childrenByParent.get(parentId) ?? [];
let output = "";
for (const view of items) {
const author = view.creator?.name ? formatAuthor(view.creator) : "unknown";
const score = view.counts?.score ?? 0;
const content = (view.comment.content ?? "").trim();
const indent = " ".repeat(depth);
output += `${indent}- **${author}** · ${score} points\n`;
if (content) {
output += `${indentBlock(content, `${indent} `)}\n`;
}
output += renderThread(view.comment.id, depth + 1);
output += "\n";
}
return output;
};
return renderThread(0, 0).trim();
}
export const handleLemmy: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
const match = parsed.pathname.match(/^\/(post|comment)\/(\d+)/);
if (!match) return null;
const kind = match[1];
const id = Number.parseInt(match[2], 10);
if (!Number.isFinite(id)) return null;
const baseUrl = parsed.origin;
const fetchedAt = new Date().toISOString();
let postId = id;
if (kind === "comment") {
const commentUrl = `${baseUrl}/api/v3/comment?id=${id}`;
const commentResult = await loadPage(commentUrl, { timeout });
if (!commentResult.ok) return null;
const commentData = parseJson<LemmyCommentResponse>(commentResult.content);
const commentView = commentData?.comment_view;
const commentPostId = commentView?.comment?.post_id;
if (!commentPostId) return null;
postId = commentPostId;
}
const postUrl = `${baseUrl}/api/v3/post?id=${postId}`;
const commentsUrl = `${baseUrl}/api/v3/comment/list?post_id=${postId}`;
const [postResult, commentsResult] = await Promise.all([
loadPage(postUrl, { timeout }),
loadPage(commentsUrl, { timeout }),
]);
if (!postResult.ok || !commentsResult.ok) return null;
const postData = parseJson<LemmyPostResponse>(postResult.content);
const postView = postData?.post_view;
if (!postView) return null;
const commentsData = parseJson<LemmyCommentListResponse>(commentsResult.content);
const comments = commentsData?.comments ?? [];
let md = `# ${postView.post.name}\n\n`;
const communityLabel = formatCommunity(postView.community);
const authorLabel = formatAuthor(postView.creator);
const score = postView.counts?.score ?? 0;
const commentCount = postView.counts?.comments ?? comments.length;
md += `**Community:** ${communityLabel} · **Author:** ${authorLabel} · **Score:** ${score} · **Comments:** ${commentCount}\n`;
if (postView.post.url) {
md += `**Link:** ${postView.post.url}\n`;
}
md += "\n";
if (postView.post.body) {
md += `---\n\n${postView.post.body}\n\n`;
}
if (comments.length > 0) {
const threadedComments = renderComments(comments);
if (threadedComments) {
md += `---\n\n## Comments\n\n${threadedComments}\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "lemmy-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Lemmy API"],
};
} catch {}
return null;
};
@@ -0,0 +1,268 @@
/**
* MusicBrainz URL handler for artists, releases, and recordings
*/
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
type MusicBrainzEntity = "artist" | "release" | "recording";
interface MusicBrainzLifeSpan {
begin?: string;
end?: string;
ended?: boolean;
}
interface MusicBrainzArtist {
id: string;
name: string;
type?: string;
country?: string;
"life-span"?: MusicBrainzLifeSpan;
}
interface MusicBrainzArtistCredit {
name?: string;
artist?: {
id?: string;
name: string;
};
}
interface MusicBrainzRecording {
id: string;
title: string;
length?: number;
"artist-credit"?: MusicBrainzArtistCredit[];
}
interface MusicBrainzTrack {
id?: string;
title?: string;
number?: string;
position?: number;
length?: number;
recording?: {
title?: string;
length?: number;
};
}
interface MusicBrainzMedium {
position?: number;
format?: string;
"track-count"?: number;
tracks?: MusicBrainzTrack[];
}
interface MusicBrainzRelease {
id: string;
title: string;
"track-count"?: number;
media?: MusicBrainzMedium[];
}
const MUSICBRAINZ_HOSTS = new Set(["musicbrainz.org", "www.musicbrainz.org"]);
const USER_AGENT = "omp-web-fetch/1.0 (https://github.com/anthropics)";
const MAX_TRACKS = 50;
function parseEntity(url: URL): { entity: MusicBrainzEntity; mbid: string } | null {
if (!MUSICBRAINZ_HOSTS.has(url.hostname)) return null;
const parts = url.pathname.split("/").filter(Boolean);
if (parts.length < 2) return null;
const entity = parts[0] as MusicBrainzEntity;
if (entity !== "artist" && entity !== "release" && entity !== "recording") return null;
const mbid = parts[1];
if (!/^[0-9a-fA-F-]{36}$/.test(mbid)) return null;
return { entity, mbid };
}
async function fetchJson<T>(apiUrl: string, timeout: number): Promise<T | null> {
const result = await loadPage(apiUrl, {
timeout,
headers: {
"User-Agent": USER_AGENT,
Accept: "application/json",
},
});
if (!result.ok) return null;
try {
return JSON.parse(result.content) as T;
} catch {
return null;
}
}
function formatLifeSpan(life: MusicBrainzLifeSpan | undefined): string | null {
if (!life) return null;
const begin = life.begin?.trim();
const end = life.end?.trim();
if (begin && end) return `${begin} - ${end}`;
if (begin && !end) return `${begin} - ${life.ended ? "ended" : "present"}`;
if (!begin && end) return `? - ${end}`;
if (life.ended !== undefined) return life.ended ? "ended" : "present";
return null;
}
function formatDurationMs(lengthMs: number | undefined): string | null {
if (!lengthMs || lengthMs <= 0) return null;
const totalSeconds = Math.round(lengthMs / 1000);
const hours = Math.floor(totalSeconds / 3600);
const minutes = Math.floor((totalSeconds % 3600) / 60);
const seconds = totalSeconds % 60;
if (hours > 0) {
return `${hours}:${minutes.toString().padStart(2, "0")}:${seconds.toString().padStart(2, "0")}`;
}
return `${minutes}:${seconds.toString().padStart(2, "0")}`;
}
function formatArtistCredits(credits: MusicBrainzArtistCredit[] | undefined): string | null {
if (!credits?.length) return null;
const names = credits
.map((credit) => credit.name || credit.artist?.name)
.filter((name): name is string => Boolean(name));
if (!names.length) return null;
return names.join(", ");
}
function formatTrack(track: MusicBrainzTrack): string {
const title = track.title || track.recording?.title || "Untitled";
const duration = formatDurationMs(track.length ?? track.recording?.length);
const number = track.number || (track.position ? String(track.position) : null);
const prefix = number ? `${number}. ` : "- ";
let line = `${prefix}${title}`;
if (duration) line += ` (${duration})`;
return line;
}
function buildMediumLabel(medium: MusicBrainzMedium, includePosition: boolean): string | null {
const parts: string[] = [];
if (includePosition && medium.position) parts.push(`Disc ${medium.position}`);
if (medium.format) parts.push(medium.format);
return parts.length ? parts.join(" - ") : null;
}
function buildArtistMarkdown(artist: MusicBrainzArtist): string {
let md = `# ${artist.name}\n\n`;
const meta: string[] = [];
if (artist.type) meta.push(`**Type**: ${artist.type}`);
if (artist.country) meta.push(`**Country**: ${artist.country}`);
const lifeSpan = formatLifeSpan(artist["life-span"]);
if (lifeSpan) meta.push(`**Life Span**: ${lifeSpan}`);
if (meta.length) md += `${meta.join("\n")}\n`;
return md;
}
function buildReleaseMarkdown(release: MusicBrainzRelease): string {
let md = `# ${release.title}\n\n`;
const media = release.media ?? [];
const totalTracks =
release["track-count"] ??
media.reduce((sum, medium) => sum + (medium["track-count"] ?? medium.tracks?.length ?? 0), 0);
if (totalTracks) {
md += `**Tracks**: ${totalTracks}\n\n`;
}
if (media.length) {
md += "## Tracks\n\n";
const includePosition = media.length > 1;
for (const medium of media) {
const label = buildMediumLabel(medium, includePosition);
if (label) md += `### ${label}\n\n`;
const tracks = medium.tracks ?? [];
if (tracks.length) {
const lines = tracks.slice(0, MAX_TRACKS).map(formatTrack).join("\n");
md += `${lines}\n\n`;
if (tracks.length > MAX_TRACKS) {
md += `_Showing first ${MAX_TRACKS} of ${tracks.length} tracks._\n\n`;
}
} else if (medium["track-count"]) {
md += `- ${medium["track-count"]} tracks (details unavailable)\n\n`;
}
}
}
return md;
}
function buildRecordingMarkdown(recording: MusicBrainzRecording): string {
let md = `# ${recording.title}\n\n`;
const meta: string[] = [];
const artists = formatArtistCredits(recording["artist-credit"]);
if (artists) meta.push(`**Artists**: ${artists}`);
const length = formatDurationMs(recording.length);
if (length) meta.push(`**Length**: ${length}`);
if (meta.length) md += `${meta.join("\n")}\n`;
return md;
}
export const handleMusicBrainz: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
const parsedEntity = parseEntity(parsed);
if (!parsedEntity) return null;
const { entity, mbid } = parsedEntity;
const fetchedAt = new Date().toISOString();
let md = "";
if (entity === "artist") {
const apiUrl = `https://musicbrainz.org/ws/2/artist/${mbid}?fmt=json&inc=url-rels`;
const artist = await fetchJson<MusicBrainzArtist>(apiUrl, timeout);
if (!artist) return null;
md = buildArtistMarkdown(artist);
} else if (entity === "release") {
const apiUrl = `https://musicbrainz.org/ws/2/release/${mbid}?fmt=json&inc=recordings`;
const release = await fetchJson<MusicBrainzRelease>(apiUrl, timeout);
if (!release) return null;
md = buildReleaseMarkdown(release);
} else {
const apiUrl = `https://musicbrainz.org/ws/2/recording/${mbid}?fmt=json`;
const recording = await fetchJson<MusicBrainzRecording>(apiUrl, timeout);
if (!recording) return null;
md = buildRecordingMarkdown(recording);
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "musicbrainz-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via MusicBrainz API"],
};
} catch {}
return null;
};
@@ -47,7 +47,7 @@ export const handleNpm: SpecialHandler = async (url: string, timeout: number): P
name: string;
version: string;
description?: string;
license?: string;
license?: string | { type: string };
homepage?: string;
repository?: { url: string } | string;
keywords?: string[];
@@ -66,7 +66,10 @@ export const handleNpm: SpecialHandler = async (url: string, timeout: number): P
if (pkg.description) md += `${pkg.description}\n\n`;
md += `**Latest:** ${pkg.version || "unknown"}`;
if (pkg.license) md += ` · **License:** ${typeof pkg.license === "string" ? pkg.license : pkg.license}`;
if (pkg.license) {
const license = typeof pkg.license === "string" ? pkg.license : (pkg.license.type ?? String(pkg.license));
md += ` · **License:** ${license}`;
}
md += "\n";
if (weeklyDownloads !== null) {
md += `**Weekly Downloads:** ${formatCount(weeklyDownloads)}\n`;
@@ -0,0 +1,263 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
interface OllamaTagDetails {
parent_model?: string;
format?: string;
family?: string;
families?: string[] | null;
parameter_size?: string;
quantization_level?: string;
}
interface OllamaTagModel {
name?: string;
model?: string;
modified_at?: string;
size?: number;
digest?: string;
details?: OllamaTagDetails;
}
interface OllamaTagsResponse {
models?: OllamaTagModel[];
}
const VALID_HOSTNAMES = new Set(["ollama.com", "www.ollama.com"]);
const RESERVED_ROOTS = new Set([
"models",
"blog",
"docs",
"download",
"cloud",
"signin",
"signout",
"search",
"api",
"terms",
"privacy",
"license",
"settings",
]);
function decodeHtmlEntities(value: string): string {
return value
.replace(/&amp;/g, "&")
.replace(/&lt;/g, "<")
.replace(/&gt;/g, ">")
.replace(/&quot;/g, '"')
.replace(/&#39;/g, "'")
.replace(/&nbsp;/g, " ");
}
function extractMetaDescription(html: string): string | null {
const patterns = [
/<meta[^>]+name=["']description["'][^>]*content=["']([^"']+)["']/i,
/<meta[^>]+property=["']og:description["'][^>]*content=["']([^"']+)["']/i,
/<meta[^>]+property=["']twitter:description["'][^>]*content=["']([^"']+)["']/i,
];
for (const pattern of patterns) {
const match = html.match(pattern);
if (match?.[1]) {
return decodeHtmlEntities(match[1].trim());
}
}
return null;
}
function extractParameterSizes(html: string): string[] {
const sizes = new Set<string>();
const pattern = /x-test-size[^>]*>([^<]+)<\/span>/gi;
let match = pattern.exec(html);
while (match) {
const raw = match[1]?.trim();
if (raw) {
sizes.add(raw.toUpperCase());
}
match = pattern.exec(html);
}
return Array.from(sizes);
}
function extractTagsFromHtml(html: string, baseRef: string): string[] {
const tags = new Set<string>();
const pattern = /href=["']\/library\/([^"']+)["']/gi;
let match = pattern.exec(html);
while (match) {
const raw = match[1]?.trim();
if (raw) {
const decoded = decodeHtmlEntities(raw);
if (decoded === baseRef || decoded.startsWith(`${baseRef}:`)) {
tags.add(decoded);
}
}
match = pattern.exec(html);
}
return Array.from(tags);
}
function formatSize(bytes: number): string {
if (bytes >= 1_000_000_000) return `${(bytes / 1_000_000_000).toFixed(1)}GB`;
if (bytes >= 1_000_000) return `${(bytes / 1_000_000).toFixed(1)}MB`;
if (bytes >= 1_000) return `${(bytes / 1_000).toFixed(1)}KB`;
return `${bytes}B`;
}
function buildModelPath(parts: string[]): string {
return parts.map((part) => encodeURIComponent(part)).join("/");
}
function parseOllamaUrl(url: string): { modelRef: string; baseRef: string; pageUrl: string } | null {
try {
const parsed = new URL(url);
if (!VALID_HOSTNAMES.has(parsed.hostname)) return null;
const parts = parsed.pathname.split("/").filter(Boolean);
if (parts.length === 0) return null;
if (parts[0] === "library" && parts.length >= 2) {
const modelRef = decodeURIComponent(parts[1]);
const baseRef = modelRef.split(":")[0] ?? modelRef;
const pageUrl = `${parsed.origin}/${buildModelPath(["library", baseRef])}`;
return { modelRef, baseRef, pageUrl };
}
if (parts.length >= 2 && !RESERVED_ROOTS.has(parts[0])) {
const namespace = decodeURIComponent(parts[0]);
const model = decodeURIComponent(parts[1]);
const modelBase = model.split(":")[0] ?? model;
const modelRef = `${namespace}/${model}`;
const baseRef = `${namespace}/${modelBase}`;
const pageUrl = `${parsed.origin}/${buildModelPath([namespace, modelBase])}`;
return { modelRef, baseRef, pageUrl };
}
} catch {}
return null;
}
function sortTags(tags: string[]): string[] {
return tags.sort((a, b) => {
const aLatest = a.endsWith(":latest");
const bLatest = b.endsWith(":latest");
if (aLatest && !bLatest) return -1;
if (!aLatest && bLatest) return 1;
return a.localeCompare(b);
});
}
function formatTagList(tags: string[], maxItems: number): string {
const limited = tags.slice(0, maxItems);
const formatted = limited.map((tag) => `\`${tag}\``).join(", ");
if (tags.length > maxItems) {
return `${formatted} (and ${tags.length - maxItems} more)`;
}
return formatted;
}
function collectParameterSizes(models: OllamaTagModel[], htmlSizes: string[]): string[] {
const sizes = new Set<string>();
for (const model of models) {
const param = model.details?.parameter_size?.trim();
if (param) sizes.add(param.toUpperCase());
}
for (const size of htmlSizes) {
sizes.add(size);
}
return Array.from(sizes);
}
export const handleOllama: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = parseOllamaUrl(url);
if (!parsed) return null;
const { modelRef, baseRef, pageUrl } = parsed;
const fetchedAt = new Date().toISOString();
const tagsUrl = "https://ollama.com/api/tags";
const [tagsResult, pageResult] = await Promise.all([
loadPage(tagsUrl, { timeout, headers: { Accept: "application/json" } }),
loadPage(pageUrl, { timeout }),
]);
let tagsData: OllamaTagsResponse | null = null;
if (tagsResult.ok) {
try {
tagsData = JSON.parse(tagsResult.content) as OllamaTagsResponse;
} catch {
tagsData = null;
}
}
const html = pageResult.ok ? pageResult.content : "";
const description = html ? extractMetaDescription(html) : null;
const htmlParameterSizes = html ? extractParameterSizes(html) : [];
const htmlTags = html ? extractTagsFromHtml(html, baseRef) : [];
const baseLower = baseRef.toLowerCase();
const models = tagsData?.models ?? [];
const matchingModels = models.filter((model) => {
const name = (model.model ?? model.name ?? "").toLowerCase();
return name === baseLower || name.startsWith(`${baseLower}:`);
});
const tagRef = modelRef.includes(":") ? modelRef : null;
const selectedTag = tagRef ? matchingModels.find((model) => (model.model ?? model.name ?? "") === tagRef) : null;
const availableTagsRaw = matchingModels
.map((model) => model.model ?? model.name ?? "")
.filter((tag) => tag.length > 0);
const availableTags = sortTags(Array.from(new Set(availableTagsRaw)));
const fallbackTags = sortTags(Array.from(new Set(htmlTags)));
const tagsToUse = availableTags.length > 0 ? availableTags : fallbackTags;
const parameterSizes = collectParameterSizes(selectedTag ? [selectedTag] : matchingModels, htmlParameterSizes);
const sizes = matchingModels
.map((model) => model.size)
.filter((size): size is number => typeof size === "number");
let sizeLine: string | null = null;
if (selectedTag?.size) {
sizeLine = formatSize(selectedTag.size);
} else if (sizes.length > 0) {
const minSize = Math.min(...sizes);
const maxSize = Math.max(...sizes);
sizeLine = minSize === maxSize ? formatSize(minSize) : `${formatSize(minSize)} - ${formatSize(maxSize)}`;
}
let md = `# ${baseRef}\n\n`;
if (description) md += `${description}\n\n`;
md += `**Model:** ${baseRef}\n`;
if (tagRef) md += `**Tag:** ${tagRef}\n`;
if (parameterSizes.length > 0) md += `**Parameters:** ${parameterSizes.join(", ")}\n`;
if (sizeLine) {
const label = sizeLine.includes(" - ") ? "Size Range" : "Size";
md += `**${label}:** ${sizeLine}\n`;
}
if (tagsToUse.length > 0) {
md += `**Available Tags:** ${formatTagList(tagsToUse, 40)}\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: pageResult.ok ? pageResult.finalUrl : url,
contentType: "text/markdown",
method: "ollama",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Ollama API"],
};
} catch {}
return null;
};
@@ -0,0 +1,115 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, loadPage } from "./types";
interface OpenVsxFileLinks {
readme?: string;
}
interface OpenVsxExtension {
name: string;
namespace: string;
version: string;
displayName?: string;
description?: string;
downloadCount?: number;
averageRating?: number;
reviewCount?: number;
repository?: string | { url?: string };
license?: string;
categories?: string[];
homepage?: string;
files?: OpenVsxFileLinks;
}
/**
* Handle Open VSX URLs via their API
*/
export const handleOpenVsx: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "open-vsx.org" && parsed.hostname !== "www.open-vsx.org") return null;
const match = parsed.pathname.match(/^\/extension\/([^/]+)\/([^/]+)(?:\/([^/]+))?\/?$/);
if (!match) return null;
const namespace = decodeURIComponent(match[1]);
const extension = decodeURIComponent(match[2]);
const version = match[3] ? decodeURIComponent(match[3]) : null;
const fetchedAt = new Date().toISOString();
const baseUrl = `https://open-vsx.org/api/${encodeURIComponent(namespace)}/${encodeURIComponent(extension)}`;
const apiUrl = version ? `${baseUrl}/${encodeURIComponent(version)}` : baseUrl;
const result = await loadPage(apiUrl, { timeout });
if (!result.ok) return null;
let data: OpenVsxExtension;
try {
data = JSON.parse(result.content);
} catch {
return null;
}
let readme: string | null = null;
const readmeUrl = data.files?.readme;
if (readmeUrl) {
try {
const readmeResult = await loadPage(readmeUrl, { timeout: Math.min(timeout, 10) });
if (readmeResult.ok) readme = readmeResult.content;
} catch {}
}
const displayName = data.displayName || data.name || `${namespace}/${extension}`;
const displayNamespace = data.namespace || namespace;
const displayVersion = data.version || version || "unknown";
const downloads = typeof data.downloadCount === "number" ? data.downloadCount : null;
const rating = typeof data.averageRating === "number" ? data.averageRating : null;
const reviews = typeof data.reviewCount === "number" ? data.reviewCount : null;
const repository = typeof data.repository === "string" ? data.repository : data.repository?.url || null;
let md = `# ${displayName}\n\n`;
if (data.description) md += `${data.description}\n\n`;
md += `**Namespace:** ${displayNamespace}\n`;
md += `**Extension:** ${data.name || extension}\n`;
md += `**Version:** ${displayVersion}`;
if (data.license) md += ` | **License:** ${data.license}`;
md += "\n";
if (downloads !== null) {
md += `**Downloads:** ${formatCount(downloads)}\n`;
}
if (rating !== null) {
const reviewSuffix = reviews !== null ? ` (${reviews} reviews)` : "";
md += `**Rating:** ${rating}${reviewSuffix}\n`;
}
if (repository) {
const cleanedRepo = repository.replace(/^git\+/, "").replace(/\.git$/, "");
md += `**Repository:** ${cleanedRepo}\n`;
}
if (data.homepage) md += `**Homepage:** ${data.homepage}\n`;
if (data.categories?.length) md += `**Categories:** ${data.categories.join(", ")}\n`;
if (readme) {
md += "\n---\n\n## README\n\n";
md += `${readme}\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "open-vsx",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Open VSX API"],
};
} catch {}
return null;
};
@@ -0,0 +1,294 @@
/**
* ORCID handler for web-fetch
*/
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
const MAX_WORKS = 50;
const ORCID_PATTERN = /\/(\d{4}-\d{4}-\d{4}-\d{3}[\dXx])(?:\/|$)/;
interface OrcidName {
"given-names"?: { value?: string };
"family-name"?: { value?: string };
"credit-name"?: { value?: string };
}
interface OrcidBiography {
content?: string;
}
interface OrcidPerson {
name?: OrcidName;
biography?: OrcidBiography;
}
interface OrcidSummaryDate {
year?: { value?: string };
month?: { value?: string };
day?: { value?: string };
}
interface OrcidOrganizationAddress {
city?: string;
region?: string;
country?: string;
}
interface OrcidOrganization {
name?: string;
address?: OrcidOrganizationAddress;
}
interface OrcidAffiliationSummary {
organization?: OrcidOrganization;
"role-title"?: string;
"department-name"?: string;
"start-date"?: OrcidSummaryDate;
"end-date"?: OrcidSummaryDate;
}
interface OrcidAffiliationGroupSummary {
"employment-summary"?: OrcidAffiliationSummary;
"education-summary"?: OrcidAffiliationSummary;
}
interface OrcidAffiliationGroup {
summaries?: OrcidAffiliationGroupSummary[];
}
interface OrcidAffiliationsContainer {
"affiliation-group"?: OrcidAffiliationGroup[];
"employment-summary"?: OrcidAffiliationSummary[];
"education-summary"?: OrcidAffiliationSummary[];
}
interface OrcidWorkTitle {
title?: { value?: string };
}
interface OrcidWorkSummary {
title?: OrcidWorkTitle;
}
interface OrcidWorkGroup {
"work-summary"?: OrcidWorkSummary[];
}
interface OrcidWorksContainer {
group?: OrcidWorkGroup[];
}
interface OrcidActivitiesSummary {
employments?: OrcidAffiliationsContainer;
educations?: OrcidAffiliationsContainer;
works?: OrcidWorksContainer;
}
interface OrcidRecord {
"orcid-identifier"?: { path?: string; uri?: string };
person?: OrcidPerson;
"activities-summary"?: OrcidActivitiesSummary;
}
function isOrcidHost(hostname: string): boolean {
return hostname === "orcid.org" || hostname === "www.orcid.org";
}
function extractOrcidId(pathname: string): string | null {
const match = pathname.match(ORCID_PATTERN);
return match?.[1] ?? null;
}
function formatName(name?: OrcidName): string | null {
const credit = name?.["credit-name"]?.value?.trim();
if (credit) return credit;
const given = name?.["given-names"]?.value?.trim();
const family = name?.["family-name"]?.value?.trim();
if (given && family) return `${given} ${family}`;
return given || family || null;
}
function formatDate(date?: OrcidSummaryDate): string | null {
const year = date?.year?.value;
if (!year) return null;
const month = date?.month?.value;
const day = date?.day?.value;
if (month && day) {
return `${year}-${month.padStart(2, "0")}-${day.padStart(2, "0")}`;
}
if (month) return `${year}-${month.padStart(2, "0")}`;
return year;
}
function collectAffiliations(
container: OrcidAffiliationsContainer | undefined,
key: "employment-summary" | "education-summary",
): OrcidAffiliationSummary[] {
const summaries: OrcidAffiliationSummary[] = [];
if (!container) return summaries;
const direct = container[key];
if (direct?.length) summaries.push(...direct);
const groups = container["affiliation-group"];
if (groups?.length) {
for (const group of groups) {
const groupSummaries = group.summaries || [];
for (const summary of groupSummaries) {
const entry = summary[key];
if (entry) summaries.push(entry);
}
}
}
return summaries;
}
function formatAffiliation(summary: OrcidAffiliationSummary): string | null {
const organization = summary.organization?.name?.trim();
const role = summary["role-title"]?.trim();
const department = summary["department-name"]?.trim();
const address = summary.organization?.address;
const locationParts = [address?.city, address?.region, address?.country].filter(Boolean) as string[];
const location = locationParts.length > 0 ? locationParts.join(", ") : null;
const start = formatDate(summary["start-date"]);
const end = formatDate(summary["end-date"]);
let dates: string | null = null;
if (start && end) {
dates = `${start} - ${end}`;
} else if (start) {
dates = `${start} - Present`;
} else if (end) {
dates = `Until ${end}`;
}
const label = organization || role || department;
if (!label) return null;
const details: string[] = [];
if (organization && role) details.push(role);
if (!organization && role && department) details.push(department);
if (organization && department) details.push(`Dept: ${department}`);
if (location) details.push(`Location: ${location}`);
if (dates) details.push(`Dates: ${dates}`);
if (details.length === 0) return label;
return `${label} (${details.join("; ")})`;
}
function collectWorkTitles(container: OrcidWorksContainer | undefined): string[] {
const titles: string[] = [];
const seen = new Set<string>();
const groups = container?.group || [];
for (const group of groups) {
const summaries = group["work-summary"] || [];
for (const summary of summaries) {
const title = summary.title?.title?.value?.trim();
if (!title || seen.has(title)) continue;
seen.add(title);
titles.push(title);
if (titles.length >= MAX_WORKS) return titles;
}
}
return titles;
}
export const handleOrcid: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!isOrcidHost(parsed.hostname)) return null;
const orcid = extractOrcidId(parsed.pathname);
if (!orcid) return null;
const fetchedAt = new Date().toISOString();
const apiUrl = `https://pub.orcid.org/v3.0/${orcid}/record`;
const result = await loadPage(apiUrl, {
timeout,
headers: { Accept: "application/json" },
});
if (!result.ok || !result.content) return null;
let record: OrcidRecord;
try {
record = JSON.parse(result.content) as OrcidRecord;
} catch {
return null;
}
const personName = formatName(record.person?.name);
const biography = record.person?.biography?.content?.trim();
const activities = record["activities-summary"];
const employments = collectAffiliations(activities?.employments, "employment-summary");
const educations = collectAffiliations(activities?.educations, "education-summary");
const works = collectWorkTitles(activities?.works);
let md = `# ${personName || "ORCID Profile"}\n\n`;
md += `**ORCID:** ${orcid}\n`;
md += `**ORCID Profile:** https://orcid.org/${orcid}\n\n`;
md += "## Biography\n\n";
md += biography ? `${biography}\n\n` : "No biography available.\n\n";
md += "## Affiliations\n\n";
let hasAffiliations = false;
if (employments.length > 0) {
hasAffiliations = true;
md += "### Employment\n\n";
for (const summary of employments) {
const line = formatAffiliation(summary);
if (line) md += `- ${line}\n`;
}
md += "\n";
}
if (educations.length > 0) {
hasAffiliations = true;
md += "### Education\n\n";
for (const summary of educations) {
const line = formatAffiliation(summary);
if (line) md += `- ${line}\n`;
}
md += "\n";
}
if (!hasAffiliations) {
md += "No affiliations available.\n\n";
}
md += "## Works\n\n";
if (works.length > 0) {
for (const title of works) {
md += `- ${title}\n`;
}
} else {
md += "No works available.\n";
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "orcid-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via ORCID Public API"],
};
} catch {
return null;
}
};
@@ -0,0 +1,120 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, htmlToBasicMarkdown, loadPage } from "./types";
interface RawgPlatformEntry {
platform?: {
name?: string;
};
}
interface RawgGenreEntry {
name?: string;
}
interface RawgGameResponse {
name?: string;
released?: string;
rating?: number;
platforms?: RawgPlatformEntry[];
genres?: RawgGenreEntry[];
description?: string;
description_raw?: string;
detail?: string;
error?: string;
}
export const handleRawg: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!isRawgHostname(parsed.hostname)) return null;
const slug = extractGameSlug(parsed.pathname);
if (!slug) return null;
const fetchedAt = new Date().toISOString();
const apiUrl = `https://api.rawg.io/api/games/${encodeURIComponent(slug)}`;
const result = await loadPage(apiUrl, { timeout, headers: { Accept: "application/json" } });
if (!result.ok) return null;
let game: RawgGameResponse;
try {
game = JSON.parse(result.content);
} catch {
return null;
}
if (requiresApiKey(game)) return null;
const title = game.name?.trim() || slug;
let md = `# ${title}\n\n`;
if (game.released) md += `**Released:** ${game.released}\n`;
if (typeof game.rating === "number" && !Number.isNaN(game.rating)) {
md += `**Rating:** ${game.rating.toFixed(2)} / 5\n`;
}
const platforms = collectNames(game.platforms?.map((entry) => entry.platform?.name));
if (platforms.length) md += `**Platforms:** ${platforms.join(", ")}\n`;
const genres = collectNames(game.genres?.map((entry) => entry.name));
if (genres.length) md += `**Genres:** ${genres.join(", ")}\n`;
md += `**RAWG:** https://rawg.io/games/${encodeURIComponent(slug)}\n`;
md += "\n";
const description = extractDescription(game);
if (description) {
md += `## Description\n\n${description}\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "rawg",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via RAWG API"],
};
} catch {}
return null;
};
function isRawgHostname(hostname: string): boolean {
return hostname === "rawg.io" || hostname === "www.rawg.io";
}
function extractGameSlug(pathname: string): string | null {
const match = pathname.match(/^\/games\/([^/?#]+)/);
if (!match) return null;
const slug = decodeURIComponent(match[1]);
return slug ? slug.trim() : null;
}
function requiresApiKey(game: RawgGameResponse): boolean {
const detail = `${game.detail ?? ""} ${game.error ?? ""}`.toLowerCase();
return detail.includes("api key") || detail.includes("key is required") || detail.includes("apikey");
}
function extractDescription(game: RawgGameResponse): string | null {
if (game.description_raw) return game.description_raw.trim();
if (!game.description) return null;
const markdown = htmlToBasicMarkdown(game.description).trim();
return markdown || null;
}
function collectNames(values?: Array<string | undefined>): string[] {
if (!values?.length) return [];
const names = new Set<string>();
for (const value of values) {
const trimmed = value?.trim();
if (trimmed) names.add(trimmed);
}
return Array.from(names);
}
@@ -0,0 +1,213 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, loadPage } from "./types";
interface SearchcodeResult {
id?: number | string;
filename?: string;
repo?: string;
language?: string;
code?: string;
lines?: number | string | Array<number | string>;
location?: string;
url?: string;
}
interface SearchcodeSearchResponse {
query?: string;
results?: SearchcodeResult[];
total?: number;
total_results?: number;
nextpage?: number;
}
const VALID_HOSTS = new Set(["searchcode.com", "www.searchcode.com"]);
function parseLineNumbers(lines: SearchcodeResult["lines"]): number[] | null {
if (typeof lines === "number" && Number.isFinite(lines)) return [lines];
if (typeof lines === "string") {
const parts = lines.split(/[,\s]+/).filter(Boolean);
const parsed = parts.map((part) => Number.parseInt(part, 10)).filter((value) => Number.isFinite(value));
return parsed.length ? parsed : null;
}
if (Array.isArray(lines)) {
const parsed = lines.map((part) => Number.parseInt(String(part), 10)).filter((value) => Number.isFinite(value));
return parsed.length ? parsed : null;
}
return null;
}
function formatLineNumbers(lines: number[] | null): string | null {
if (!lines || lines.length === 0) return null;
if (lines.length <= 10) return lines.join(", ");
const min = Math.min(...lines);
const max = Math.max(...lines);
return `${min}-${max} (${lines.length} lines)`;
}
function formatCodeBlock(
code: string | undefined,
language: string | undefined,
lines: number[] | null,
): string | null {
if (!code) return null;
const normalized = code.replace(/\r\n/g, "\n").trimEnd();
const codeLines = normalized.split("\n");
const languageTag = typeof language === "string" ? language.trim().toLowerCase() : "";
let displayLines = codeLines;
if (lines && lines.length === codeLines.length) {
displayLines = codeLines.map((line, index) => `${lines[index]}: ${line}`);
}
const fence = languageTag ? languageTag : "";
return `\n\n\`\`\`${fence}\n${displayLines.join("\n")}\n\`\`\`\n`;
}
export const handleSearchcode: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!VALID_HOSTS.has(parsed.hostname)) return null;
const fetchedAt = new Date().toISOString();
const viewMatch = parsed.pathname.match(/^\/codesearch\/view\/([^/?#]+)/);
if (viewMatch) {
const id = viewMatch[1];
const apiUrl = `https://searchcode.com/api/result/${encodeURIComponent(id)}/`;
const result = await loadPage(apiUrl, { timeout, headers: { Accept: "application/json" } });
if (!result.ok) return null;
let data: SearchcodeResult;
try {
data = JSON.parse(result.content) as SearchcodeResult;
} catch {
return null;
}
const filename = data.filename || data.location || `Result ${id}`;
const lineNumbers = parseLineNumbers(data.lines);
const formattedLines = formatLineNumbers(lineNumbers);
const viewUrl = data.url || `https://searchcode.com/codesearch/view/${id}`;
const snippetBlock = formatCodeBlock(data.code, data.language, lineNumbers);
let md = `# ${filename}\n\n`;
md += `## Description\n\n`;
md += "Code snippet from searchcode.com.\n\n";
md += `## Metadata\n\n`;
if (data.repo) md += `**Repository:** ${data.repo}\n`;
if (data.language) md += `**Language:** ${data.language}\n`;
if (data.filename) md += `**File:** ${data.filename}\n`;
if (data.location) md += `**Location:** ${data.location}\n`;
if (formattedLines) md += `**Lines:** ${formattedLines}\n`;
md += `**Result ID:** ${id}\n`;
md += `**URL:** ${viewUrl}\n`;
md += `\n## Snippet`;
if (snippetBlock) {
md += snippetBlock;
} else {
md += "\n\n_No snippet available._\n";
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "searchcode",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via searchcode API"],
};
}
const query = parsed.searchParams.get("q");
const isSearchPage =
parsed.pathname === "/" || parsed.pathname === "/codesearch" || parsed.pathname === "/codesearch/";
if (!query || !isSearchPage) return null;
const pageRaw = parsed.searchParams.get("p") ?? parsed.searchParams.get("page");
const pageNumber = pageRaw ? Number.parseInt(pageRaw, 10) : 0;
const page = Number.isFinite(pageNumber) && pageNumber >= 0 ? pageNumber : 0;
const apiUrl = `https://searchcode.com/api/codesearch_I/?q=${encodeURIComponent(query)}&p=${page}`;
const result = await loadPage(apiUrl, { timeout, headers: { Accept: "application/json" } });
if (!result.ok) return null;
let data: SearchcodeSearchResponse;
try {
data = JSON.parse(result.content) as SearchcodeSearchResponse;
} catch {
return null;
}
const results = Array.isArray(data.results) ? data.results : [];
const total =
typeof data.total === "number"
? data.total
: typeof data.total_results === "number"
? data.total_results
: null;
let md = `# Searchcode Results\n\n`;
md += `## Description\n\n`;
md += `Search results for \`${query}\` on searchcode.com.\n\n`;
md += `## Metadata\n\n`;
md += `**Query:** \`${query}\`\n`;
md += `**Page:** ${page}\n`;
if (total !== null) md += `**Total Results:** ${formatCount(total)}\n`;
md += `**Result Count:** ${results.length}\n`;
if (typeof data.nextpage === "number") md += `**Next Page:** ${data.nextpage}\n`;
md += `\n## Results\n\n`;
if (results.length === 0) {
md += "_No results found._\n";
} else {
const maxResults = 10;
for (const resultItem of results.slice(0, maxResults)) {
const id = resultItem.id !== undefined ? String(resultItem.id) : null;
const filename = resultItem.filename || resultItem.location || "Result";
const lineNumbers = parseLineNumbers(resultItem.lines);
const formattedLines = formatLineNumbers(lineNumbers);
const viewUrl = resultItem.url || (id ? `https://searchcode.com/codesearch/view/${id}` : null);
const snippetBlock = formatCodeBlock(resultItem.code, resultItem.language, lineNumbers);
md += `### ${filename}\n\n`;
if (resultItem.repo) md += `**Repository:** ${resultItem.repo}\n`;
if (resultItem.language) md += `**Language:** ${resultItem.language}\n`;
if (resultItem.filename) md += `**File:** ${resultItem.filename}\n`;
if (resultItem.location) md += `**Location:** ${resultItem.location}\n`;
if (formattedLines) md += `**Lines:** ${formattedLines}\n`;
if (viewUrl) md += `**URL:** ${viewUrl}\n`;
if (snippetBlock) {
md += `${snippetBlock}\n`;
}
md += "\n";
}
if (results.length > maxResults) {
md += `\n_Only showing first ${maxResults} results._\n`;
}
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "searchcode",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via searchcode API"],
};
} catch {}
return null;
};
@@ -0,0 +1,195 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, loadPage } from "./types";
interface SnapcraftPublisher {
"display-name"?: string;
username?: string;
id?: string;
validation?: string;
}
interface SnapcraftChannel {
name?: string;
track?: string;
risk?: string;
branch?: string | null;
architecture?: string;
"released-at"?: string;
}
interface SnapcraftDownload {
size?: number;
url?: string;
"sha3-384"?: string;
}
interface SnapcraftChannelMapEntry {
channel?: SnapcraftChannel;
version?: string;
revision?: number | string;
download?: SnapcraftDownload;
type?: string;
"created-at"?: string;
}
interface SnapcraftSnap {
name?: string;
title?: string;
summary?: string;
description?: string;
publisher?: SnapcraftPublisher;
version?: string;
confinement?: string;
base?: string;
downloads?: number;
download?: number;
}
interface SnapcraftResponse {
name?: string;
title?: string;
summary?: string;
description?: string;
publisher?: SnapcraftPublisher;
version?: string;
confinement?: string;
base?: string;
downloads?: number;
download?: number;
snap?: SnapcraftSnap;
"channel-map"?: SnapcraftChannelMapEntry[];
}
function formatPublisher(publisher?: SnapcraftPublisher): string | null {
if (!publisher) return null;
const displayName = publisher["display-name"] ?? publisher.username ?? publisher.id;
if (!displayName) return null;
if (publisher.username && displayName !== publisher.username) {
return `${displayName} (@${publisher.username})`;
}
return displayName;
}
function formatChannelName(channel?: SnapcraftChannel): string | null {
if (!channel) return null;
if (channel.name?.includes("/")) return channel.name;
if (channel.track && channel.risk) {
const branch = channel.branch ? `/${channel.branch}` : "";
return `${channel.track}/${channel.risk}${branch}`;
}
return channel.name ?? null;
}
function pickVersionFromChannels(entries: SnapcraftChannelMapEntry[]): string | undefined {
const stable = entries.find((entry) => entry.channel?.risk === "stable" && entry.version);
if (stable?.version) return stable.version;
const first = entries.find((entry) => entry.version);
return first?.version;
}
function extractDownloads(snapInfo: SnapcraftSnap | SnapcraftResponse, data: SnapcraftResponse): number | null {
const candidates = [snapInfo.downloads, snapInfo.download, data.downloads, data.download];
for (const value of candidates) {
if (typeof value === "number" && Number.isFinite(value)) return value;
}
return null;
}
export const handleSnapcraft: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "snapcraft.io" && parsed.hostname !== "www.snapcraft.io") return null;
const installMatch = parsed.pathname.match(/^\/install\/([^/]+)\/?$/);
const directMatch = parsed.pathname.match(/^\/([^/]+)\/?$/);
if (!installMatch && !directMatch) return null;
const snapName = decodeURIComponent((installMatch ?? directMatch)![1]);
const fetchedAt = new Date().toISOString();
const apiUrl = `https://api.snapcraft.io/v2/snaps/info/${encodeURIComponent(snapName)}`;
const result = await loadPage(apiUrl, {
timeout,
headers: {
Accept: "application/json",
"Snap-Device-Series": "16",
},
});
if (!result.ok) return null;
let data: SnapcraftResponse;
try {
data = JSON.parse(result.content) as SnapcraftResponse;
} catch {
return null;
}
const snapInfo = data.snap ?? data;
const name = snapInfo.title ?? snapInfo.name ?? data.name ?? snapName;
const summary = snapInfo.summary ?? data.summary;
const description = snapInfo.description ?? data.description;
const publisher = formatPublisher(snapInfo.publisher ?? data.publisher);
const confinement = snapInfo.confinement ?? data.confinement;
const base = snapInfo.base ?? data.base;
const channelMap = data["channel-map"] ?? [];
let version = snapInfo.version ?? data.version;
if (!version && channelMap.length > 0) {
version = pickVersionFromChannels(channelMap);
}
const downloads = extractDownloads(snapInfo, data);
const channels = new Map<string, { version?: string; architectures: Set<string> }>();
for (const entry of channelMap) {
const channelName = formatChannelName(entry.channel);
if (!channelName) continue;
const existing = channels.get(channelName) ?? { architectures: new Set<string>() };
if (!existing.version && entry.version) existing.version = entry.version;
if (entry.channel?.architecture) existing.architectures.add(entry.channel.architecture);
channels.set(channelName, existing);
}
let md = `# ${name}\n\n`;
if (summary) md += `${summary}\n\n`;
md += `**Version:** ${version ?? "unknown"}`;
if (confinement) md += ` · **Confinement:** ${confinement}`;
if (base) md += ` · **Base:** ${base}`;
md += "\n";
if (publisher) md += `**Publisher:** ${publisher}\n`;
if (downloads !== null) md += `**Downloads:** ${formatCount(downloads)}\n`;
md += "\n";
if (channels.size > 0) {
md += "## Channels\n\n";
const sortedChannels = Array.from(channels.entries()).sort((a, b) => a[0].localeCompare(b[0]));
for (const [channelName, info] of sortedChannels) {
const arches = Array.from(info.architectures).sort();
const versionSuffix = info.version ? `: ${info.version}` : "";
const archSuffix = arches.length > 0 ? ` (${arches.join(", ")})` : "";
md += `- ${channelName}${versionSuffix}${archSuffix}\n`;
}
md += "\n";
}
const descriptionText = description ?? summary;
if (descriptionText) {
md += `## Description\n\n${descriptionText}\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "snapcraft",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via Snapcraft API"],
};
} catch {}
return null;
};
@@ -0,0 +1,353 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, loadPage } from "./types";
const GRAPHQL_ENDPOINT = "https://sourcegraph.com/.api/graphql";
const GRAPHQL_HEADERS = {
Accept: "application/json",
"Content-Type": "application/json",
};
type SourcegraphTarget =
| { type: "search"; query: string }
| { type: "repo"; repoName: string; rev?: string }
| { type: "file"; repoName: string; rev?: string; filePath: string };
interface SourcegraphRepository {
name: string;
url: string;
description?: string | null;
defaultBranch?: { name: string } | null;
}
interface RepoQueryData {
repository?: SourcegraphRepository | null;
}
interface RepoFileQueryData {
repository?:
| (SourcegraphRepository & {
commit?: {
blob?: { content?: string | null } | null;
} | null;
})
| null;
}
interface SearchQueryData {
search?: {
results?: {
results?: SearchResultItem[] | null;
matchCount?: number | null;
limitHit?: boolean | null;
} | null;
} | null;
}
interface FileMatchResult {
__typename: "FileMatch";
repository?: { name?: string | null; url?: string | null } | null;
file?: { path?: string | null; url?: string | null } | null;
lineMatches?: Array<{ preview?: string | null; lineNumber?: number | null }> | null;
}
interface RepositoryResult {
__typename: "Repository";
name?: string | null;
url?: string | null;
}
type SearchResultItem = FileMatchResult | RepositoryResult | { __typename: string };
const REPO_QUERY = `query Repo($name: String!) {
repository(name: $name) {
name
url
description
defaultBranch {
name
}
}
}`;
const REPO_FILE_QUERY = `query RepoFile($name: String!, $path: String!, $rev: String!) {
repository(name: $name) {
name
url
description
defaultBranch {
name
}
commit(rev: $rev) {
blob(path: $path) {
content
}
}
}
}`;
const SEARCH_QUERY = `query Search($query: String!) {
search(query: $query, version: V2) {
results {
results {
__typename
... on FileMatch {
repository {
name
url
}
file {
path
url
}
lineMatches {
preview
lineNumber
}
}
... on Repository {
name
url
}
}
matchCount
limitHit
}
}
}`;
function parseSourcegraphUrl(url: string): SourcegraphTarget | null {
try {
const parsed = new URL(url);
if (parsed.hostname !== "sourcegraph.com" && parsed.hostname !== "www.sourcegraph.com") return null;
if (parsed.pathname.startsWith("/search")) {
const query = parsed.searchParams.get("q")?.trim();
if (!query) return null;
return { type: "search", query };
}
const parts = parsed.pathname
.split("/")
.filter(Boolean)
.map((part) => decodeURIComponent(part));
if (parts.length < 3) return null;
const hyphenIndex = parts.indexOf("-");
const repoParts = hyphenIndex === -1 ? parts : parts.slice(0, hyphenIndex);
if (repoParts.length < 3) return null;
const lastRepoPart = repoParts[repoParts.length - 1];
const atIndex = lastRepoPart.indexOf("@");
let rev: string | undefined;
let repoTail = lastRepoPart;
if (atIndex > 0) {
repoTail = lastRepoPart.slice(0, atIndex);
rev = lastRepoPart.slice(atIndex + 1) || undefined;
}
repoParts[repoParts.length - 1] = repoTail;
const repoName = repoParts.join("/");
if (hyphenIndex !== -1 && parts[hyphenIndex + 1] === "blob") {
const filePath = parts.slice(hyphenIndex + 2).join("/");
if (!filePath) return null;
return { type: "file", repoName, rev, filePath };
}
return { type: "repo", repoName, rev };
} catch {
return null;
}
}
function safeParseJson<T>(content: string): T | null {
try {
return JSON.parse(content) as T;
} catch {
return null;
}
}
async function fetchGraphql<T>(query: string, variables: Record<string, unknown>, timeout: number): Promise<T | null> {
const body = JSON.stringify({ query, variables });
const result = await loadPage(GRAPHQL_ENDPOINT, {
timeout,
headers: GRAPHQL_HEADERS,
method: "POST",
body,
});
if (!result.ok) return null;
const parsed = safeParseJson<{ data?: T; errors?: unknown }>(result.content);
if (!parsed?.data) return null;
if (Array.isArray(parsed.errors) && parsed.errors.length > 0) return null;
return parsed.data;
}
function isFileMatchResult(result: SearchResultItem): result is FileMatchResult {
return result.__typename === "FileMatch";
}
function isRepositoryResult(result: SearchResultItem): result is RepositoryResult {
return result.__typename === "Repository";
}
function formatRepoMarkdown(repo: SourcegraphRepository): string {
let md = `# ${repo.name}\n\n`;
if (repo.description) md += `${repo.description}\n\n`;
md += `**URL:** ${repo.url}\n`;
if (repo.defaultBranch?.name) md += `**Default branch:** ${repo.defaultBranch.name}\n`;
return md;
}
async function renderRepo(repoName: string, timeout: number): Promise<{ content: string; ok: boolean }> {
const data = await fetchGraphql<RepoQueryData>(REPO_QUERY, { name: repoName }, timeout);
if (!data?.repository) return { content: "", ok: false };
return { content: formatRepoMarkdown(data.repository), ok: true };
}
async function renderFile(
repoName: string,
filePath: string,
rev: string,
timeout: number,
): Promise<{ content: string; ok: boolean }> {
const data = await fetchGraphql<RepoFileQueryData>(
REPO_FILE_QUERY,
{ name: repoName, path: filePath, rev },
timeout,
);
const repo = data?.repository;
const content = repo?.commit?.blob?.content ?? null;
if (!repo || content === null) return { content: "", ok: false };
let md = `${formatRepoMarkdown(repo)}\n`;
md += `**Path:** ${filePath}\n`;
md += `**Revision:** ${rev}\n\n`;
md += `---\n\n## File\n\n`;
md += "```text\n";
md += `${content}\n`;
md += "```\n";
return { content: md, ok: true };
}
async function renderSearch(query: string, timeout: number): Promise<{ content: string; ok: boolean }> {
const data = await fetchGraphql<SearchQueryData>(SEARCH_QUERY, { query }, timeout);
const resultsData = data?.search?.results;
if (!resultsData) return { content: "", ok: false };
const results = resultsData.results ?? [];
let md = "# Sourcegraph Search\n\n";
md += `**Query:** \`${query}\`\n`;
if (typeof resultsData?.matchCount === "number") {
md += `**Matches:** ${resultsData.matchCount}\n`;
}
if (typeof resultsData?.limitHit === "boolean") {
md += `**Limit hit:** ${resultsData.limitHit ? "yes" : "no"}\n`;
}
md += "\n";
if (!results || results.length === 0) {
md += "_No results._\n";
return { content: md, ok: true };
}
const maxResults = 10;
md += "## Results\n\n";
for (const result of results.slice(0, maxResults)) {
if (isFileMatchResult(result)) {
const repoName = result.repository?.name ?? "unknown";
const filePath = result.file?.path ?? "unknown";
md += `### ${repoName}/${filePath}\n\n`;
if (result.repository?.url) md += `**Repository:** ${result.repository.url}\n`;
if (result.file?.url) md += `**File:** ${result.file.url}\n`;
const lineMatches = result.lineMatches ?? [];
if (lineMatches.length > 0) {
md += "\n```text\n";
for (const line of lineMatches.slice(0, 5)) {
const preview = (line.preview ?? "").replace(/\n/g, " ").trim();
const lineNumber = line.lineNumber ?? 0;
md += `L${lineNumber}: ${preview}\n`;
}
md += "```\n\n";
}
continue;
}
if (isRepositoryResult(result)) {
const name = result.name ?? "unknown";
md += `### ${name}\n\n`;
if (result.url) md += `**Repository:** ${result.url}\n`;
md += "\n";
}
}
if (results.length > maxResults) {
md += `... and ${results.length - maxResults} more results\n`;
}
return { content: md, ok: true };
}
export const handleSourcegraph: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const target = parseSourcegraphUrl(url);
if (!target) return null;
const fetchedAt = new Date().toISOString();
const notes = ["Fetched via Sourcegraph GraphQL API"];
switch (target.type) {
case "search": {
const result = await renderSearch(target.query, timeout);
if (!result.ok) return null;
const output = finalizeOutput(result.content);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "sourcegraph-search",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
case "file": {
const rev = target.rev ?? "HEAD";
const result = await renderFile(target.repoName, target.filePath, rev, timeout);
if (!result.ok) return null;
const output = finalizeOutput(result.content);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "sourcegraph-file",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
case "repo": {
const result = await renderRepo(target.repoName, timeout);
if (!result.ok) return null;
const output = finalizeOutput(result.content);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "sourcegraph-repo",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes,
};
}
}
} catch {}
return null;
};
@@ -0,0 +1,116 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, htmlToBasicMarkdown, loadPage } from "./types";
interface SpdxCrossRef {
url?: string;
isValid?: boolean;
isLive?: boolean;
match?: string;
order?: number;
}
interface SpdxLicense {
licenseId: string;
name: string;
isOsiApproved?: boolean;
isFsfLibre?: boolean;
licenseText?: string;
licenseTextHtml?: string;
seeAlso?: string[];
crossRef?: SpdxCrossRef[];
comment?: string;
licenseComments?: string;
}
function formatYesNo(value?: boolean): string {
if (value === true) return "Yes";
if (value === false) return "No";
return "Unknown";
}
function collectCrossReferences(license: SpdxLicense): string[] {
const ordered = (license.crossRef ?? [])
.filter((ref) => ref.url)
.sort((a, b) => (a.order ?? 0) - (b.order ?? 0))
.map((ref) => ref.url as string);
const seeAlso = (license.seeAlso ?? []).filter((url) => url);
const combined = [...ordered, ...seeAlso];
return combined.filter((url, index) => combined.indexOf(url) === index);
}
/**
* Handle SPDX license URLs via SPDX JSON API
*/
export const handleSpdx: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "spdx.org" && parsed.hostname !== "www.spdx.org") return null;
const match = parsed.pathname.match(/^\/licenses\/([^/]+?)(?:\.html)?\/?$/i);
if (!match) return null;
const licenseId = decodeURIComponent(match[1]);
if (!licenseId) return null;
const fetchedAt = new Date().toISOString();
const apiUrl = `https://spdx.org/licenses/${encodeURIComponent(licenseId)}.json`;
const result = await loadPage(apiUrl, {
timeout,
headers: { Accept: "application/json" },
});
if (!result.ok) return null;
let license: SpdxLicense;
try {
license = JSON.parse(result.content);
} catch {
return null;
}
const title = license.name || license.licenseId || licenseId;
let md = `# ${title}\n\n`;
md += `**License ID:** ${license.licenseId ? `\`${license.licenseId}\`` : `\`${licenseId}\``}\n`;
md += `**OSI Approved:** ${formatYesNo(license.isOsiApproved)}\n`;
md += `**FSF Libre:** ${formatYesNo(license.isFsfLibre)}\n`;
const description = license.licenseComments ?? license.comment;
if (description) {
md += `\n## Description\n\n${description}\n`;
}
const crossReferences = collectCrossReferences(license);
if (crossReferences.length) {
md += `\n## Cross References\n\n`;
for (const ref of crossReferences) {
md += `- ${ref}\n`;
}
}
const licenseText = license.licenseText
? license.licenseText
: license.licenseTextHtml
? htmlToBasicMarkdown(license.licenseTextHtml)
: null;
if (licenseText) {
md += `\n## License Text\n\n\`\`\`\n${licenseText}\n\`\`\`\n`;
}
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "spdx-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via SPDX license API"],
};
} catch {}
return null;
};
@@ -16,6 +16,61 @@ export interface RenderResult {
export type SpecialHandler = (url: string, timeout: number) => Promise<RenderResult | null>;
export const MAX_OUTPUT_CHARS = 500_000;
const MAX_BYTES = 50 * 1024 * 1024;
const USER_AGENTS = [
"curl/8.0",
"Mozilla/5.0 (compatible; TextBot/1.0)",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
];
export interface RequestSignal {
signal: AbortSignal;
cleanup: () => void;
}
export function createRequestSignal(timeoutMs: number, signal?: AbortSignal): RequestSignal {
const controller = new AbortController();
let timeoutId: ReturnType<typeof setTimeout> | undefined = setTimeout(() => controller.abort(), timeoutMs);
const abortHandler = () => controller.abort();
if (signal) {
if (signal.aborted) {
clearTimeout(timeoutId);
timeoutId = undefined;
controller.abort();
} else {
signal.addEventListener("abort", abortHandler, { once: true });
}
}
const cleanup = () => {
if (timeoutId !== undefined) {
clearTimeout(timeoutId);
timeoutId = undefined;
}
if (signal) {
signal.removeEventListener("abort", abortHandler);
}
};
return { signal: controller.signal, cleanup };
}
function isBotBlocked(status: number, content: string): boolean {
if (status === 403 || status === 503) {
const lower = content.toLowerCase();
return (
lower.includes("cloudflare") ||
lower.includes("captcha") ||
lower.includes("challenge") ||
lower.includes("blocked") ||
lower.includes("access denied") ||
lower.includes("bot detection")
);
}
return false;
}
/**
* Truncate and cleanup output
@@ -29,30 +84,41 @@ export function finalizeOutput(content: string): { content: string; truncated: b
};
}
export interface LoadPageOptions {
timeout?: number;
headers?: Record<string, string>;
method?: string;
body?: string;
maxBytes?: number;
signal?: AbortSignal;
}
export interface LoadPageResult {
content: string;
contentType: string;
finalUrl: string;
ok: boolean;
status?: number;
}
/**
* Fetch a page with timeout and size limit
*/
export async function loadPage(
url: string,
options: { timeout?: number; headers?: Record<string, string>; maxBytes?: number } = {},
): Promise<{ content: string; contentType: string; finalUrl: string; ok: boolean; status?: number }> {
const { timeout = 20, headers = {}, maxBytes = 50 * 1024 * 1024 } = options;
export async function loadPage(url: string, options: LoadPageOptions = {}): Promise<LoadPageResult> {
const { timeout = 20, headers = {}, maxBytes = MAX_BYTES, signal, method = "GET", body } = options;
const userAgents = [
"curl/8.0",
"Mozilla/5.0 (compatible; TextBot/1.0)",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
];
for (let attempt = 0; attempt < USER_AGENTS.length; attempt++) {
if (signal?.aborted) {
return { content: "", contentType: "", finalUrl: url, ok: false };
}
for (let attempt = 0; attempt < userAgents.length; attempt++) {
const userAgent = userAgents[attempt];
const userAgent = USER_AGENTS[attempt];
const { signal: requestSignal, cleanup } = createRequestSignal(timeout * 1000, signal);
try {
const controller = new AbortController();
const timeoutId = setTimeout(() => controller.abort(), timeout * 1000);
const response = await fetch(url, {
signal: controller.signal,
const requestInit: RequestInit = {
signal: requestSignal,
method,
headers: {
"User-Agent": userAgent,
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
@@ -60,9 +126,13 @@ export async function loadPage(
...headers,
},
redirect: "follow",
});
};
clearTimeout(timeoutId);
if (body !== undefined) {
requestInit.body = body;
}
const response = await fetch(url, requestInit);
const contentType = response.headers.get("content-type")?.split(";")[0]?.trim().toLowerCase() ?? "";
const finalUrl = response.url;
@@ -91,12 +161,8 @@ export async function loadPage(
const decoder = new TextDecoder();
const content = decoder.decode(Buffer.concat(chunks));
// Check if blocked
if ((response.status === 403 || response.status === 503) && attempt < userAgents.length - 1) {
const lower = content.toLowerCase();
if (lower.includes("cloudflare") || lower.includes("captcha") || lower.includes("blocked")) {
continue;
}
if (isBotBlocked(response.status, content) && attempt < USER_AGENTS.length - 1) {
continue;
}
if (!response.ok) {
@@ -105,9 +171,14 @@ export async function loadPage(
return { content, contentType, finalUrl, ok: true, status: response.status };
} catch (_err) {
if (attempt === userAgents.length - 1) {
if (signal?.aborted) {
return { content: "", contentType: "", finalUrl: url, ok: false };
}
if (attempt === USER_AGENTS.length - 1) {
return { content: "", contentType: "", finalUrl: url, ok: false };
}
} finally {
cleanup();
}
}
@@ -0,0 +1,161 @@
import { tmpdir } from "node:os";
import * as path from "node:path";
import { ensureTool } from "../../../utils/tools-manager";
import { createRequestSignal } from "./types";
const MAX_BYTES = 50 * 1024 * 1024; // 50MB for binary files
interface ExecResult {
stdout: string;
stderr: string;
ok: boolean;
exitCode: number;
}
type SpawnSyncOptions = NonNullable<Parameters<typeof Bun.spawnSync>[1]>;
function exec(cmd: string, args: string[], options?: { timeout?: number; input?: string | Buffer }): ExecResult {
const stdin = (options?.input ?? "ignore") as SpawnSyncOptions["stdin"];
const result = Bun.spawnSync([cmd, ...args], {
stdin,
stdout: "pipe",
stderr: "pipe",
});
return {
stdout: result.stdout?.toString() ?? "",
stderr: result.stderr?.toString() ?? "",
ok: result.exitCode === 0,
exitCode: result.exitCode ?? -1,
};
}
export interface ConvertResult {
content: string;
ok: boolean;
error?: string;
}
export interface BinaryFetchResult {
buffer: Buffer;
contentType: string;
contentDisposition?: string;
ok: boolean;
status?: number;
error?: string;
}
export async function convertWithMarkitdown(
content: Buffer,
extensionHint: string,
timeout: number,
signal?: AbortSignal,
): Promise<ConvertResult> {
if (signal?.aborted) {
return { content: "", ok: false, error: "aborted" };
}
const markitdown = await ensureTool("markitdown", true);
if (!markitdown) {
return { content: "", ok: false, error: "markitdown not available" };
}
// Write to temp file with extension hint
const ext = extensionHint || ".bin";
const tmpDir = tmpdir();
const tmpFile = path.join(tmpDir, `omp-convert-${Date.now()}${ext}`);
if (content.length > MAX_BYTES) {
return { content: "", ok: false, error: `content exceeds ${MAX_BYTES} bytes` };
}
try {
await Bun.write(tmpFile, content);
const result = exec(markitdown, [tmpFile], { timeout });
if (!result.ok) {
const stderr = result.stderr.trim();
return {
content: result.stdout,
ok: false,
error: stderr.length > 0 ? stderr : `markitdown failed (exit ${result.exitCode})`,
};
}
return { content: result.stdout, ok: true };
} finally {
try {
await Bun.$`rm ${tmpFile}`.quiet();
} catch {}
}
}
export async function fetchBinary(url: string, timeout: number, signal?: AbortSignal): Promise<BinaryFetchResult> {
if (signal?.aborted) {
return { buffer: Buffer.alloc(0), contentType: "", ok: false, error: "aborted" };
}
const { signal: requestSignal, cleanup } = createRequestSignal(timeout * 1000, signal);
try {
const response = await fetch(url, {
signal: requestSignal,
headers: {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/131.0.0.0",
},
redirect: "follow",
});
const contentType = response.headers.get("content-type") ?? "";
const contentDisposition = response.headers.get("content-disposition") ?? undefined;
if (!response.ok) {
return {
buffer: Buffer.alloc(0),
contentType,
contentDisposition,
ok: false,
status: response.status,
error: `status ${response.status}`,
};
}
const contentLength = response.headers.get("content-length");
if (contentLength) {
const size = Number.parseInt(contentLength, 10);
if (Number.isFinite(size) && size > MAX_BYTES) {
return {
buffer: Buffer.alloc(0),
contentType,
contentDisposition,
ok: false,
status: response.status,
error: `content-length ${size} exceeds ${MAX_BYTES}`,
};
}
}
const buffer = Buffer.from(await response.arrayBuffer());
if (buffer.length > MAX_BYTES) {
return {
buffer: Buffer.alloc(0),
contentType,
contentDisposition,
ok: false,
status: response.status,
error: `response exceeds ${MAX_BYTES} bytes`,
};
}
return { buffer, contentType, contentDisposition, ok: true, status: response.status };
} catch (err) {
if (signal?.aborted) {
return { buffer: Buffer.alloc(0), contentType: "", ok: false, error: "aborted" };
}
return {
buffer: Buffer.alloc(0),
contentType: "",
ok: false,
error: `request failed: ${String(err)}`,
};
} finally {
cleanup();
}
}
@@ -0,0 +1,193 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, formatCount, loadPage } from "./types";
interface MarketplaceProperty {
key?: string;
value?: string;
}
interface MarketplaceVersion {
version?: string;
properties?: MarketplaceProperty[];
}
interface MarketplaceStatistic {
statisticName?: string;
value?: number;
}
interface MarketplacePublisher {
publisherName?: string;
displayName?: string;
}
interface MarketplaceExtension {
extensionName?: string;
displayName?: string;
shortDescription?: string;
description?: string;
publisher?: MarketplacePublisher;
versions?: MarketplaceVersion[];
statistics?: MarketplaceStatistic[];
categories?: string[];
tags?: string[];
properties?: MarketplaceProperty[];
}
interface MarketplaceResponse {
results?: Array<{ extensions?: MarketplaceExtension[] }>;
}
const MARKETPLACE_HOSTS = new Set(["marketplace.visualstudio.com", "www.marketplace.visualstudio.com"]);
function getItemName(parsed: URL): string | null {
if (!parsed.pathname.startsWith("/items")) return null;
const itemName = parsed.searchParams.get("itemName");
if (!itemName) return null;
const decoded = decodeURIComponent(itemName);
if (!decoded.includes(".")) return null;
return decoded;
}
function toStatMap(stats: MarketplaceStatistic[] | undefined): Map<string, number> {
const map = new Map<string, number>();
if (!stats) return map;
for (const stat of stats) {
if (!stat.statisticName || typeof stat.value !== "number") continue;
map.set(stat.statisticName.trim().toLowerCase(), stat.value);
}
return map;
}
function formatRating(averageRating?: number, ratingCount?: number): string | null {
if (averageRating === undefined && ratingCount === undefined) return null;
if (averageRating !== undefined) {
const formatted = averageRating.toFixed(2).replace(/\.0+$/, "").replace(/\.$/, "");
if (ratingCount !== undefined) {
return `${formatted} (${formatCount(ratingCount)} ratings)`;
}
return formatted;
}
if (ratingCount !== undefined) {
return `${formatCount(ratingCount)} ratings`;
}
return null;
}
function extractRepoLink(properties: MarketplaceProperty[] | undefined): string | null {
if (!properties) return null;
for (const prop of properties) {
const key = prop.key?.trim().toLowerCase();
const value = prop.value?.trim();
if (!key || !value) continue;
if (!value.startsWith("http")) continue;
if (key.includes("links.source") || key.includes("repository")) return value;
}
for (const prop of properties) {
const key = prop.key?.trim().toLowerCase();
const value = prop.value?.trim();
if (!key || !value) continue;
if (!value.startsWith("http")) continue;
if (key === "source" || key.endsWith(".source")) return value;
}
return null;
}
/**
* Handle VS Code Marketplace URLs via extension query API
*/
export const handleVscodeMarketplace: SpecialHandler = async (
url: string,
timeout: number,
): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (!MARKETPLACE_HOSTS.has(parsed.hostname)) return null;
const itemName = getItemName(parsed);
if (!itemName) return null;
const [publisherFromUrl, ...nameParts] = itemName.split(".");
const extensionFromUrl = nameParts.join(".");
const fetchedAt = new Date().toISOString();
const apiUrl = "https://marketplace.visualstudio.com/_apis/public/gallery/extensionquery";
const payload = JSON.stringify({
filters: [
{
criteria: [{ filterType: 7, value: itemName }],
},
],
flags: 950,
});
const result = await loadPage(apiUrl, {
timeout,
method: "POST",
body: payload,
headers: {
"Content-Type": "application/json",
Accept: "application/json;api-version=7.2-preview.1",
},
});
if (!result.ok) return null;
let data: MarketplaceResponse;
try {
data = JSON.parse(result.content) as MarketplaceResponse;
} catch {
return null;
}
const extension = data.results?.[0]?.extensions?.[0];
if (!extension) return null;
const extensionName = extension.extensionName ?? extensionFromUrl;
const displayName = extension.displayName ?? extensionName ?? itemName;
const description = extension.shortDescription ?? extension.description;
const publisherName = extension.publisher?.publisherName ?? publisherFromUrl;
const publisherDisplayName = extension.publisher?.displayName;
const publisherLabel =
publisherDisplayName && publisherName && publisherDisplayName !== publisherName
? `${publisherDisplayName} (${publisherName})`
: (publisherDisplayName ?? publisherName);
const version = extension.versions?.[0]?.version;
const statMap = toStatMap(extension.statistics);
const installs = statMap.get("install") ?? statMap.get("installs");
const averageRating = statMap.get("averagerating");
const ratingCount = statMap.get("ratingcount");
const ratingLabel = formatRating(averageRating, ratingCount);
const repoLink = extractRepoLink(extension.versions?.[0]?.properties) ?? extractRepoLink(extension.properties);
const identifier = publisherName && extensionName ? `${publisherName}.${extensionName}` : itemName;
let md = `# ${displayName}\n\n`;
if (description) md += `${description}\n\n`;
md += `**Identifier:** ${identifier}\n`;
if (publisherLabel) md += `**Publisher:** ${publisherLabel}\n`;
if (version) md += `**Version:** ${version}\n`;
if (installs !== undefined) md += `**Installs:** ${formatCount(installs)}\n`;
if (ratingLabel) md += `**Rating:** ${ratingLabel}\n`;
if (extension.categories?.length) md += `**Categories:** ${extension.categories.join(", ")}\n`;
if (extension.tags?.length) md += `**Tags:** ${extension.tags.join(", ")}\n`;
if (repoLink) md += `**Repository:** ${repoLink}\n`;
const output = finalizeOutput(md);
return {
url,
finalUrl: url,
contentType: "text/markdown",
method: "vscode-marketplace",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via VS Code Marketplace API"],
};
} catch {}
return null;
};
@@ -0,0 +1,159 @@
import type { RenderResult, SpecialHandler } from "./types";
import { finalizeOutput, htmlToBasicMarkdown, loadPage } from "./types";
type JsonRecord = Record<string, unknown>;
function asRecord(value: unknown): JsonRecord | null {
if (!value || typeof value !== "object" || Array.isArray(value)) return null;
return value as JsonRecord;
}
function getString(record: JsonRecord | null, key: string): string | undefined {
if (!record) return undefined;
const value = record[key];
return typeof value === "string" ? value : undefined;
}
function getRecord(record: JsonRecord | null, key: string): JsonRecord | null {
if (!record) return null;
return asRecord(record[key]);
}
function getArray(record: JsonRecord | null, key: string): unknown[] | undefined {
if (!record) return undefined;
const value = record[key];
return Array.isArray(value) ? value : undefined;
}
function extractShortname(pathname: string): string | null {
const trimmed = pathname.replace(/\/+$/g, "");
const segments = trimmed.split("/").filter(Boolean);
if (segments.length < 2 || segments[0] !== "TR") return null;
if (segments.length === 2) {
const shortname = segments[1];
if (/^\d{4}$/.test(shortname)) return null;
return decodeURIComponent(shortname);
}
if (segments.length >= 3 && /^\d{4}$/.test(segments[1])) {
const version = segments[2];
const match = version.match(/^[A-Za-z]+-(.+)-\d{8}$/);
if (match?.[1]) return decodeURIComponent(match[1]);
}
return null;
}
function normalizeStatus(status?: string): { code?: string; label?: string } {
if (!status) return {};
const lower = status.toLowerCase();
if (lower.includes("working draft")) return { code: "WD", label: status };
if (lower.includes("candidate recommendation")) return { code: "CR", label: status };
if (lower.includes("proposed recommendation")) return { code: "PR", label: status };
if (lower.includes("recommendation")) return { code: "REC", label: status };
return { label: status };
}
function extractEditors(editorsPayload: JsonRecord | null): string[] {
const links = getRecord(editorsPayload, "_links");
const editors = getArray(links, "editors") ?? [];
const names: string[] = [];
for (const entry of editors) {
const record = asRecord(entry);
const title = getString(record, "title");
if (title) names.push(title);
}
return names;
}
export const handleW3c: SpecialHandler = async (url: string, timeout: number): Promise<RenderResult | null> => {
try {
const parsed = new URL(url);
if (parsed.hostname !== "www.w3.org" && parsed.hostname !== "w3.org") return null;
const shortname = extractShortname(parsed.pathname);
if (!shortname) return null;
const fetchedAt = new Date().toISOString();
const specUrl = `https://api.w3.org/specifications/${encodeURIComponent(shortname)}`;
const latestUrl = `https://api.w3.org/specifications/${encodeURIComponent(shortname)}/versions/latest`;
const [specResult, latestResult] = await Promise.all([
loadPage(specUrl, { timeout, headers: { Accept: "application/json" } }),
loadPage(latestUrl, { timeout, headers: { Accept: "application/json" } }),
]);
if (!specResult.ok || !latestResult.ok) return null;
const specPayload = asRecord(JSON.parse(specResult.content));
const latestPayload = asRecord(JSON.parse(latestResult.content));
if (!specPayload || !latestPayload) return null;
const title = getString(specPayload, "title");
const shortnameValue = getString(specPayload, "shortname") ?? shortname;
const description = getString(specPayload, "description") ?? getString(specPayload, "abstract");
const abstract = description ? htmlToBasicMarkdown(description) : undefined;
const latestVersionUrl =
getString(latestPayload, "uri") ??
getString(latestPayload, "shortlink") ??
getString(specPayload, "shortlink");
const latestStatus = getString(latestPayload, "status");
const normalizedStatus = normalizeStatus(latestStatus);
const specLinks = getRecord(specPayload, "_links");
const historyUrl = getString(getRecord(specLinks, "version-history"), "href");
const latestLinks = getRecord(latestPayload, "_links");
const editorsUrl = getString(getRecord(latestLinks, "editors"), "href");
let editors: string[] = [];
if (editorsUrl) {
const editorsResult = await loadPage(editorsUrl, { timeout: Math.min(timeout, 10) });
if (editorsResult.ok) {
try {
const editorsPayload = asRecord(JSON.parse(editorsResult.content));
editors = editorsPayload ? extractEditors(editorsPayload) : [];
} catch {}
}
}
let md = `# ${title ?? shortnameValue}\n\n`;
if (abstract) md += `## Abstract\n\n${abstract}\n\n`;
md += "## Metadata\n\n";
md += `**Shortname:** ${shortnameValue}\n`;
if (normalizedStatus.code) {
md += `**Status:** ${normalizedStatus.code}`;
if (normalizedStatus.label) md += ` (${normalizedStatus.label})`;
md += "\n";
} else if (normalizedStatus.label) {
md += `**Status:** ${normalizedStatus.label}\n`;
}
if (editors.length) md += `**Editors:** ${editors.join(", ")}\n`;
if (latestVersionUrl) md += `**Latest Version:** ${latestVersionUrl}\n`;
if (historyUrl) md += `**History:** ${historyUrl}\n`;
const output = finalizeOutput(md);
return {
url,
finalUrl: latestVersionUrl ?? url,
contentType: "text/markdown",
method: "w3c-api",
content: output.content,
fetchedAt,
truncated: output.truncated,
notes: ["Fetched via W3C API"],
};
} catch {}
return null;
};