refactor(tools): consolidated fetch into read tool with URL caching
- Consolidated fetch tool into read tool with URL reading capability and caching support. - Removed standalone fetch tool from all agent prompts and CLI documentation. - Extended read tool schema with timeout and raw parameters for URL fetch control. - Added URL caching mechanism to prevent redundant network requests during read operations. - Refactored fetch module from class-based tool to standalone executeReadUrl function. - Updated read tool documentation to describe multi-purpose capabilities including web pages, GitHub, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, blogs, and feeds.
This commit is contained in:
@@ -1,6 +1,26 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Breaking Changes
|
||||
|
||||
- Removed standalone `fetch` tool; URL fetching is now integrated into the `read` tool
|
||||
|
||||
### Added
|
||||
|
||||
- Added URL reading capability to `read` tool with support for web pages, GitHub issues, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, technical blogs, RSS/Atom feeds, and JSON endpoints
|
||||
- Added `offset` and `limit` parameter support for paginating cached URL fetch results
|
||||
- Added URL caching mechanism to avoid redundant network requests when reading the same URL multiple times
|
||||
|
||||
### Changed
|
||||
|
||||
- Renamed `fetch.enabled` setting to `Read URLs` with updated description to reflect integration with read tool
|
||||
- Updated `read` tool to accept `timeout` and `raw` parameters for URL handling
|
||||
- Updated `read` tool to support `file://` URLs for local file paths
|
||||
- Removed `fetch` tool from agent tool lists (explore, librarian, oracle, plan, reviewer agents)
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `read` tool to properly handle `file://` URL scheme by converting to filesystem paths
|
||||
|
||||
## [13.17.5] - 2026-04-01
|
||||
### Added
|
||||
|
||||
@@ -262,7 +262,6 @@ ${chalk.bold("Available Tools (default-enabled unless noted):")}
|
||||
browser - Browser automation (Puppeteer)
|
||||
task - Launch sub-agents for parallel tasks
|
||||
todo_write - Manage todo/task lists
|
||||
fetch - Fetch and process URLs
|
||||
web_search - Search the web
|
||||
ask - Ask user questions (interactive mode only)
|
||||
|
||||
|
||||
@@ -1207,7 +1207,7 @@ export const SETTINGS_SCHEMA = {
|
||||
"fetch.enabled": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
ui: { tab: "tools", label: "Fetch", description: "Enable the fetch tool for URL fetching" },
|
||||
ui: { tab: "tools", label: "Read URLs", description: "Allow the read tool to fetch and process URLs" },
|
||||
},
|
||||
|
||||
"github.enabled": {
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
/**
|
||||
* Types for the internal URL routing system.
|
||||
*
|
||||
* Internal URLs (agent://, artifact://, memory://, skill://, rule://, mcp://, pi://, local://) are resolved by tools like fetch and read,
|
||||
* Internal URLs (agent://, artifact://, memory://, skill://, rule://, mcp://, pi://, local://) are resolved by tools like read,
|
||||
* providing access to agent outputs and server resources without exposing filesystem paths.
|
||||
*/
|
||||
|
||||
|
||||
@@ -106,7 +106,6 @@ export function mapToolKind(toolName: string): ToolKind {
|
||||
case "find":
|
||||
case "ast_grep":
|
||||
return "search";
|
||||
case "fetch":
|
||||
case "web_search":
|
||||
return "fetch";
|
||||
case "todo_write":
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: explore
|
||||
description: Fast read-only codebase scout returning compressed context for handoff
|
||||
tools: read, grep, find, fetch, web_search
|
||||
tools: read, grep, find, web_search
|
||||
model: pi/smol
|
||||
thinking-level: med
|
||||
output:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: librarian
|
||||
description: Researches external libraries and APIs by reading source code. Returns definitive, source-verified answers.
|
||||
tools: read, grep, find, bash, lsp, web_search, fetch, ast_grep
|
||||
tools: read, grep, find, bash, lsp, web_search, ast_grep
|
||||
model: pi/smol
|
||||
thinking-level: minimal
|
||||
output:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: oracle
|
||||
description: Deep reasoning advisor for debugging dead ends, architecture decisions, and second opinions. Read-only.
|
||||
tools: read, grep, find, bash, lsp, fetch, web_search, ast_grep
|
||||
tools: read, grep, find, bash, lsp, web_search, ast_grep
|
||||
spawns: explore
|
||||
model: pi/slow
|
||||
thinking-level: high
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: plan
|
||||
description: Software architect for complex multi-file architectural decisions. NOT for simple tasks, single-file changes, or tasks completable in <5 tool calls.
|
||||
tools: read, grep, find, bash, lsp, fetch, web_search, ast_grep
|
||||
tools: read, grep, find, bash, lsp, web_search, ast_grep
|
||||
spawns: explore
|
||||
model: pi/plan, pi/slow
|
||||
thinking-level: high
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: "Code review specialist for quality/security analysis"
|
||||
tools: read, grep, find, bash, lsp, fetch, web_search, ast_grep, report_finding
|
||||
tools: read, grep, find, bash, lsp, web_search, ast_grep, report_finding
|
||||
spawns: explore
|
||||
model: pi/slow
|
||||
thinking-level: high
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
Retrieves content from a URL and returns it in a clean, readable format.
|
||||
|
||||
<instruction>
|
||||
- Extract information from web pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, technical blogs, RSS/Atom feeds, JSON endpoints
|
||||
- Read PDF or DOCX files hosted at a URL
|
||||
- Use `raw: true` for untouched HTML or debugging
|
||||
</instruction>
|
||||
|
||||
<output>
|
||||
Returns processed, readable content. HTML transformed to remove boilerplate. PDF/DOCX converted to text. JSON returned formatted. With `raw: true`, returns untransformed HTML.
|
||||
</output>
|
||||
@@ -1,31 +1,40 @@
|
||||
Reads files from local filesystem, supported archives, or internal URLs.
|
||||
Reads the content at the specified path or URL.
|
||||
|
||||
<instruction>
|
||||
- Reads up to {{DEFAULT_LIMIT}} lines default
|
||||
The `read` tool is a multi-purpose tool that can be used to inspect all kinds of files and URLs.
|
||||
- You **MUST** parallelize reads when exploring related files
|
||||
|
||||
# Filesystem
|
||||
- Reads up to {{DEFAULT_LIMIT}} lines by default
|
||||
- Use `offset` and `limit` for large files; max {{DEFAULT_MAX_LINES}} lines per call
|
||||
{{#if IS_HASHLINE_MODE}}
|
||||
- Filesystem output is CID prefixed: `LINE#ID:content`
|
||||
- If reading from FS, result will be prefixed with anchors: `41#ZZ:def alpha():`
|
||||
{{else}}
|
||||
{{#if IS_LINE_NUMBER_MODE}}
|
||||
- Filesystem output is line-number-prefixed
|
||||
{{#if IS_LINE_NUMBER_MODE}}
|
||||
- If reading from FS, result will be prefixed with line numbers: `41:def alpha():`
|
||||
{{/if}}
|
||||
{{/if}}
|
||||
{{/if}}
|
||||
- Supports images (PNG, JPG) and PDFs
|
||||
- For directories, returns formatted listing with modification times
|
||||
- Supports `.tar`, `.tar.gz`, `.tgz`, and `.zip` archives
|
||||
- Use `archive.ext:path/inside/archive` to read or list archive contents
|
||||
- Parallelize reads when exploring related files
|
||||
</instruction>
|
||||
|
||||
<output>
|
||||
- Returns file content as text; images return visual content; PDFs return extracted text; archive roots behave like directories
|
||||
- Missing files: returns closest filename matches for correction
|
||||
</output>
|
||||
# Inspection
|
||||
When used with a PDF, Word, PowerPoint, Excel, RTF, EPUB, or Jupyter notebook file, the tool will return the extracted text.
|
||||
It can also be used to inspect images.
|
||||
|
||||
# Directories & Archives
|
||||
When used against a directory, or an archive root, the tool will return a list of directory entries within.
|
||||
- Formats: `.tar`, `.tar.gz`, `.tgz`, and `.zip`.
|
||||
- Use `archive.ext:path/inside/archive` to read or list archive contents
|
||||
|
||||
# URLs
|
||||
- Extract information from web pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, technical blogs, RSS/Atom feeds, JSON endpoints
|
||||
- `raw: true` for untouched HTML or debugging
|
||||
- `timeout` to override the default request timeout
|
||||
</instruction>
|
||||
|
||||
<critical>
|
||||
- You **MUST** use `read` instead of bash for ALL file reading: `cat`, `head`, `tail`, `less`, `more` are FORBIDDEN.
|
||||
- You **MUST** use `read(path="dir/")` instead of `ls dir/` for directory listings.
|
||||
- You **MUST** use `read(path="archive.zip:path/to/file")` instead of shelling out to `tar` or `unzip` for supported archive reads.
|
||||
- You **MUST** always include the `path` parameter — NEVER call `read` with empty arguments `{}`.
|
||||
- You **MUST** use `read` instead of `ls` for directory listings.
|
||||
- You **MUST** use `read` instead of shelling out to `tar` or `unzip` for supported archive reads.
|
||||
- You **MUST** always include the `path` parameter, NEVER call `read` with empty arguments `{}`.
|
||||
- When reading specific line ranges, use `offset` and `limit`: `read(path="file", offset=50, limit=100)` not `cat -n file | sed`.
|
||||
- You **MAY** use `offset` and `limit` with URL reads; the tool will paginate the cached fetched output.
|
||||
</critical>
|
||||
|
||||
@@ -75,7 +75,7 @@ export class ArtifactManager {
|
||||
/**
|
||||
* Allocate a new artifact path and ID without writing content.
|
||||
*
|
||||
* @param toolType Tool name for file extension (e.g., "bash", "fetch")
|
||||
* @param toolType Tool name for file extension (e.g., "bash", "read")
|
||||
*/
|
||||
async allocatePath(toolType: string): Promise<{ id: string; path: string }> {
|
||||
await this.#ensureDir();
|
||||
@@ -88,7 +88,7 @@ export class ArtifactManager {
|
||||
* Save content as an artifact and return the artifact ID.
|
||||
*
|
||||
* @param content Full content to save
|
||||
* @param toolType Tool name for file extension (e.g., "bash", "fetch")
|
||||
* @param toolType Tool name for file extension (e.g., "bash", "read")
|
||||
* @returns Artifact ID (numeric string)
|
||||
*/
|
||||
async save(content: string, toolType: string): Promise<string> {
|
||||
|
||||
@@ -496,7 +496,7 @@ export class TaskTool implements AgentTool<TaskSchema, TaskToolDetails, Theme> {
|
||||
}
|
||||
|
||||
const planModeState = this.session.getPlanModeState?.();
|
||||
const planModeTools = ["read", "grep", "find", "ls", "lsp", "fetch", "web_search"];
|
||||
const planModeTools = ["read", "grep", "find", "ls", "lsp", "web_search"];
|
||||
const effectiveAgent: typeof agent = planModeState?.enabled
|
||||
? {
|
||||
...agent,
|
||||
|
||||
@@ -1,16 +1,15 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import { htmlToMarkdown } from "@oh-my-pi/pi-natives";
|
||||
import { type Component, Text } from "@oh-my-pi/pi-tui";
|
||||
import { ptree, truncate } from "@oh-my-pi/pi-utils";
|
||||
import { type Static, Type } from "@sinclair/typebox";
|
||||
import { parseHTML } from "linkedom";
|
||||
import { renderPromptTemplate } from "../config/prompt-templates";
|
||||
import type { Settings } from "../config/settings";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
import { type Theme, theme } from "../modes/theme/theme";
|
||||
import fetchDescription from "../prompts/tools/fetch.md" with { type: "text" };
|
||||
import type { ToolSession } from "../sdk";
|
||||
import { DEFAULT_MAX_BYTES, truncateHead } from "../session/streaming-output";
|
||||
import { renderStatusLine } from "../tui";
|
||||
import { CachedOutputBlock } from "../tui/output-block";
|
||||
@@ -21,7 +20,6 @@ import { specialHandlers } from "../web/scrapers";
|
||||
import type { RenderResult } from "../web/scrapers/types";
|
||||
import { finalizeOutput, loadPage, MAX_OUTPUT_CHARS } from "../web/scrapers/types";
|
||||
import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils";
|
||||
import type { ToolSession } from ".";
|
||||
import { applyListLimit } from "./list-limit";
|
||||
import { formatStyledArtifactReference, type OutputMeta } from "./output-meta";
|
||||
import { formatExpandHint, getDomain } from "./render-utils";
|
||||
@@ -135,6 +133,10 @@ function normalizeUrl(url: string): string {
|
||||
return url;
|
||||
}
|
||||
|
||||
export function isReadableUrlPath(value: string): boolean {
|
||||
return /^https?:\/\//i.test(value) || /^www\./i.test(value);
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize MIME type (lowercase, strip charset/params)
|
||||
*/
|
||||
@@ -1082,13 +1084,8 @@ async function renderUrl(
|
||||
// Tool Definition
|
||||
// =============================================================================
|
||||
|
||||
const fetchSchema = Type.Object({
|
||||
url: Type.String({ description: "URL to fetch" }),
|
||||
timeout: Type.Optional(Type.Number({ description: "Timeout in seconds (default: 20)" })),
|
||||
raw: Type.Optional(Type.Boolean({ description: "Return raw HTML without transforms" })),
|
||||
});
|
||||
|
||||
export interface FetchToolDetails {
|
||||
export interface ReadUrlToolDetails {
|
||||
kind: "url";
|
||||
url: string;
|
||||
finalUrl: string;
|
||||
contentType: string;
|
||||
@@ -1098,96 +1095,180 @@ export interface FetchToolDetails {
|
||||
meta?: OutputMeta;
|
||||
}
|
||||
|
||||
export class FetchTool implements AgentTool<typeof fetchSchema, FetchToolDetails> {
|
||||
readonly name = "fetch";
|
||||
readonly label = "Fetch";
|
||||
readonly description: string;
|
||||
readonly parameters = fetchSchema;
|
||||
readonly strict = true;
|
||||
interface ReadUrlCacheEntry {
|
||||
artifactId?: string;
|
||||
details: ReadUrlToolDetails;
|
||||
image?: FetchImagePayload;
|
||||
output: string;
|
||||
}
|
||||
|
||||
constructor(private readonly session: ToolSession) {
|
||||
this.description = renderPromptTemplate(fetchDescription);
|
||||
const readUrlCache = new Map<string, ReadUrlCacheEntry>();
|
||||
|
||||
function getReadUrlCacheKey(session: ToolSession, requestedUrl: string, raw: boolean): string {
|
||||
const scope = session.getSessionFile() ?? session.cwd;
|
||||
return `${scope}::${raw ? "raw" : "rendered"}::${normalizeUrl(requestedUrl)}`;
|
||||
}
|
||||
|
||||
async function readArtifactOutput(session: ToolSession, artifactId: string): Promise<string | null> {
|
||||
const artifactsDir = session.getArtifactsDir?.();
|
||||
if (!artifactsDir) return null;
|
||||
|
||||
try {
|
||||
const files = await fs.readdir(artifactsDir);
|
||||
const match = files.find(file => file.startsWith(`${artifactId}.`));
|
||||
if (!match) return null;
|
||||
return await Bun.file(path.join(artifactsDir, match)).text();
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function materializeReadUrlCacheEntry(
|
||||
session: ToolSession,
|
||||
entry: ReadUrlCacheEntry,
|
||||
): Promise<ReadUrlCacheEntry | null> {
|
||||
if (entry.artifactId) {
|
||||
const artifactOutput = await readArtifactOutput(session, entry.artifactId);
|
||||
if (artifactOutput !== null) {
|
||||
return { ...entry, output: artifactOutput };
|
||||
}
|
||||
}
|
||||
|
||||
async execute(
|
||||
_toolCallId: string,
|
||||
params: Static<typeof fetchSchema>,
|
||||
signal?: AbortSignal,
|
||||
_onUpdate?: AgentToolUpdateCallback<FetchToolDetails>,
|
||||
_context?: AgentToolContext,
|
||||
): Promise<AgentToolResult<FetchToolDetails>> {
|
||||
const { url, timeout: rawTimeout = 20, raw = false } = params;
|
||||
return entry.output.length > 0 ? entry : null;
|
||||
}
|
||||
|
||||
// Clamp to valid range (seconds)
|
||||
const effectiveTimeout = clampTimeout("fetch", rawTimeout);
|
||||
async function persistReadUrlArtifact(session: ToolSession, output: string): Promise<string | undefined> {
|
||||
const { path: artifactPath, id } = (await session.allocateOutputArtifact?.("read")) ?? {};
|
||||
if (!artifactPath) return undefined;
|
||||
await Bun.write(artifactPath, output);
|
||||
return id;
|
||||
}
|
||||
|
||||
if (signal?.aborted) {
|
||||
throw new ToolAbortError();
|
||||
}
|
||||
async function ensureReadUrlCacheArtifact(session: ToolSession, entry: ReadUrlCacheEntry): Promise<ReadUrlCacheEntry> {
|
||||
if (entry.artifactId) return entry;
|
||||
const artifactId = await persistReadUrlArtifact(session, entry.output);
|
||||
return artifactId ? { ...entry, artifactId } : entry;
|
||||
}
|
||||
|
||||
const result = await renderUrl(url, effectiveTimeout, raw, this.session.settings, signal);
|
||||
const truncation = truncateHead(result.content, {
|
||||
maxBytes: DEFAULT_MAX_BYTES,
|
||||
maxLines: FETCH_DEFAULT_MAX_LINES,
|
||||
});
|
||||
const needsArtifact = truncation.truncated;
|
||||
let artifactId: string | undefined;
|
||||
function cacheReadUrlEntry(session: ToolSession, requestedUrl: string, raw: boolean, entry: ReadUrlCacheEntry): void {
|
||||
readUrlCache.set(getReadUrlCacheKey(session, requestedUrl, raw), entry);
|
||||
readUrlCache.set(getReadUrlCacheKey(session, entry.details.finalUrl, raw), entry);
|
||||
}
|
||||
|
||||
const buildOutput = (content: string): string => {
|
||||
let output = "";
|
||||
output += `URL: ${result.finalUrl}\n`;
|
||||
output += `Content-Type: ${result.contentType}\n`;
|
||||
output += `Method: ${result.method}\n`;
|
||||
if (result.notes.length > 0) {
|
||||
output += `Notes: ${result.notes.join("; ")}\n`;
|
||||
}
|
||||
output += `\n---\n\n`;
|
||||
output += content;
|
||||
return output;
|
||||
};
|
||||
async function buildReadUrlCacheEntry(
|
||||
session: ToolSession,
|
||||
params: { path: string; timeout?: number; raw?: boolean },
|
||||
signal?: AbortSignal,
|
||||
options?: { ensureArtifact?: boolean },
|
||||
): Promise<ReadUrlCacheEntry> {
|
||||
const { path: url, timeout: rawTimeout = 20, raw = false } = params;
|
||||
|
||||
if (needsArtifact) {
|
||||
const { path: artifactPath, id } = (await this.session.allocateOutputArtifact?.("fetch")) ?? {};
|
||||
if (artifactPath) {
|
||||
await Bun.write(artifactPath, buildOutput(result.content));
|
||||
artifactId = id;
|
||||
}
|
||||
}
|
||||
const effectiveTimeout = clampTimeout("fetch", rawTimeout);
|
||||
|
||||
const output = buildOutput(needsArtifact ? truncation.content : result.content);
|
||||
if (signal?.aborted) {
|
||||
throw new ToolAbortError();
|
||||
}
|
||||
|
||||
const details: FetchToolDetails = {
|
||||
const result = await renderUrl(url, effectiveTimeout, raw, session.settings, signal);
|
||||
const output = buildUrlReadOutput(result, result.content);
|
||||
const artifactId = options?.ensureArtifact ? await persistReadUrlArtifact(session, output) : undefined;
|
||||
|
||||
return {
|
||||
artifactId,
|
||||
details: {
|
||||
kind: "url",
|
||||
url: result.url,
|
||||
finalUrl: result.finalUrl,
|
||||
contentType: result.contentType,
|
||||
method: result.method,
|
||||
truncated: Boolean(result.truncated || needsArtifact),
|
||||
truncated: Boolean(result.truncated),
|
||||
notes: result.notes,
|
||||
};
|
||||
},
|
||||
image: result.image,
|
||||
output,
|
||||
};
|
||||
}
|
||||
|
||||
const contentBlocks: Array<TextContent | ImageContent> = [{ type: "text", text: output }];
|
||||
if (result.image) {
|
||||
contentBlocks.push({ type: "image", data: result.image.data, mimeType: result.image.mimeType });
|
||||
export async function loadReadUrlCacheEntry(
|
||||
session: ToolSession,
|
||||
params: { path: string; timeout?: number; raw?: boolean },
|
||||
signal?: AbortSignal,
|
||||
options?: { ensureArtifact?: boolean; preferCached?: boolean },
|
||||
): Promise<ReadUrlCacheEntry> {
|
||||
const raw = params.raw ?? false;
|
||||
const cached = readUrlCache.get(getReadUrlCacheKey(session, params.path, raw));
|
||||
if (options?.preferCached && cached) {
|
||||
const prepared = options.ensureArtifact ? await ensureReadUrlCacheArtifact(session, cached) : cached;
|
||||
const materialized = await materializeReadUrlCacheEntry(session, prepared);
|
||||
if (materialized) {
|
||||
cacheReadUrlEntry(session, params.path, raw, materialized);
|
||||
return materialized;
|
||||
}
|
||||
|
||||
const resultBuilder = toolResult(details).content(contentBlocks).sourceUrl(result.finalUrl);
|
||||
if (needsArtifact) {
|
||||
resultBuilder.truncation(truncation, { direction: "head", artifactId });
|
||||
} else if (result.truncated) {
|
||||
const outputLines = result.content.split("\n").length;
|
||||
const outputBytes = Buffer.byteLength(result.content, "utf-8");
|
||||
const totalBytes = Math.max(outputBytes + 1, MAX_OUTPUT_CHARS + 1);
|
||||
const totalLines = outputLines + 1;
|
||||
resultBuilder.truncationFromText(result.content, {
|
||||
direction: "tail",
|
||||
totalLines,
|
||||
totalBytes,
|
||||
maxBytes: MAX_OUTPUT_CHARS,
|
||||
});
|
||||
}
|
||||
|
||||
return resultBuilder.done();
|
||||
}
|
||||
|
||||
const fresh = await buildReadUrlCacheEntry(session, params, signal, {
|
||||
ensureArtifact: options?.ensureArtifact,
|
||||
});
|
||||
cacheReadUrlEntry(session, params.path, raw, fresh);
|
||||
return fresh;
|
||||
}
|
||||
|
||||
function buildUrlReadOutput(result: FetchRenderResult, content: string): string {
|
||||
let output = "";
|
||||
output += `URL: ${result.finalUrl}\n`;
|
||||
output += `Content-Type: ${result.contentType}\n`;
|
||||
output += `Method: ${result.method}\n`;
|
||||
if (result.notes.length > 0) {
|
||||
output += `Notes: ${result.notes.join("; ")}\n`;
|
||||
}
|
||||
output += `\n---\n\n`;
|
||||
output += content;
|
||||
return output;
|
||||
}
|
||||
|
||||
export async function executeReadUrl(
|
||||
session: ToolSession,
|
||||
params: { path: string; timeout?: number; raw?: boolean },
|
||||
signal?: AbortSignal,
|
||||
): Promise<AgentToolResult<ReadUrlToolDetails>> {
|
||||
let cacheEntry = await loadReadUrlCacheEntry(session, params, signal, { preferCached: true });
|
||||
const truncation = truncateHead(cacheEntry.output, {
|
||||
maxBytes: DEFAULT_MAX_BYTES,
|
||||
maxLines: FETCH_DEFAULT_MAX_LINES,
|
||||
});
|
||||
const needsArtifact = truncation.truncated;
|
||||
if (needsArtifact && !cacheEntry.artifactId) {
|
||||
cacheEntry = await ensureReadUrlCacheArtifact(session, cacheEntry);
|
||||
cacheReadUrlEntry(session, params.path, params.raw ?? false, cacheEntry);
|
||||
}
|
||||
const output = needsArtifact ? truncation.content : cacheEntry.output;
|
||||
const details: ReadUrlToolDetails = {
|
||||
...cacheEntry.details,
|
||||
truncated: Boolean(cacheEntry.details.truncated || needsArtifact),
|
||||
};
|
||||
|
||||
const contentBlocks: Array<TextContent | ImageContent> = [{ type: "text", text: output }];
|
||||
if (cacheEntry.image) {
|
||||
contentBlocks.push({ type: "image", data: cacheEntry.image.data, mimeType: cacheEntry.image.mimeType });
|
||||
}
|
||||
|
||||
const resultBuilder = toolResult(details).content(contentBlocks).sourceUrl(details.finalUrl);
|
||||
if (needsArtifact) {
|
||||
resultBuilder.truncation(truncation, { direction: "head", artifactId: cacheEntry.artifactId });
|
||||
} else if (cacheEntry.details.truncated) {
|
||||
const outputLines = cacheEntry.output.split("\n").length;
|
||||
const outputBytes = Buffer.byteLength(cacheEntry.output, "utf-8");
|
||||
const totalBytes = Math.max(outputBytes + 1, MAX_OUTPUT_CHARS + 1);
|
||||
const totalLines = outputLines + 1;
|
||||
resultBuilder.truncationFromText(cacheEntry.output, {
|
||||
direction: "tail",
|
||||
totalLines,
|
||||
totalBytes,
|
||||
maxBytes: MAX_OUTPUT_CHARS,
|
||||
});
|
||||
}
|
||||
|
||||
return resultBuilder.done();
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -1199,26 +1280,26 @@ function countNonEmptyLines(text: string): number {
|
||||
return text.split("\n").filter(l => l.trim()).length;
|
||||
}
|
||||
|
||||
/** Render fetch call (URL preview) */
|
||||
export function renderFetchCall(
|
||||
args: { url?: string; timeout?: number; raw?: boolean },
|
||||
/** Render URL read call (URL preview) */
|
||||
export function renderReadUrlCall(
|
||||
args: { path?: string; url?: string; timeout?: number; raw?: boolean },
|
||||
_options: RenderResultOptions,
|
||||
uiTheme: Theme = theme,
|
||||
): Component {
|
||||
const url = args.url ?? "";
|
||||
const url = args.path ?? args.url ?? "";
|
||||
const domain = getDomain(url);
|
||||
const path = truncate(url.replace(/^https?:\/\/[^/]+/, ""), 50, "\u2026");
|
||||
const description = `${domain}${path ? ` ${path}` : ""}`.trim();
|
||||
const meta: string[] = [];
|
||||
if (args.raw) meta.push("raw");
|
||||
if (args.timeout !== undefined) meta.push(`timeout:${args.timeout}s`);
|
||||
const text = renderStatusLine({ icon: "pending", title: "Fetch", description, meta }, uiTheme);
|
||||
const text = renderStatusLine({ icon: "pending", title: "Read", description, meta }, uiTheme);
|
||||
return new Text(text, 0, 0);
|
||||
}
|
||||
|
||||
/** Render fetch result with tree-based layout */
|
||||
export function renderFetchResult(
|
||||
result: { content: Array<{ type: string; text?: string }>; details?: FetchToolDetails },
|
||||
/** Render URL read result with tree-based layout */
|
||||
export function renderReadUrlResult(
|
||||
result: { content: Array<{ type: string; text?: string }>; details?: ReadUrlToolDetails },
|
||||
options: RenderResultOptions,
|
||||
uiTheme: Theme = theme,
|
||||
): Component {
|
||||
@@ -1238,7 +1319,7 @@ export function renderFetchResult(
|
||||
const header = renderStatusLine(
|
||||
{
|
||||
icon: truncated ? "warning" : "success",
|
||||
title: "Fetch",
|
||||
title: "Read",
|
||||
description: `${domain}${path ? ` ${path}` : ""}`,
|
||||
},
|
||||
uiTheme,
|
||||
@@ -1316,9 +1397,3 @@ export function renderFetchResult(
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export const fetchToolRenderer = {
|
||||
renderCall: renderFetchCall,
|
||||
renderResult: renderFetchResult,
|
||||
mergeCallAndResult: true,
|
||||
};
|
||||
|
||||
@@ -26,7 +26,6 @@ import { CalculatorTool } from "./calculator";
|
||||
import { CancelJobTool } from "./cancel-job";
|
||||
import { type CheckpointState, CheckpointTool, RewindTool } from "./checkpoint";
|
||||
import { ExitPlanModeTool } from "./exit-plan-mode";
|
||||
import { FetchTool } from "./fetch";
|
||||
import { FindTool } from "./find";
|
||||
import {
|
||||
GhIssueViewTool,
|
||||
@@ -73,7 +72,6 @@ export * from "./calculator";
|
||||
export * from "./cancel-job";
|
||||
export * from "./checkpoint";
|
||||
export * from "./exit-plan-mode";
|
||||
export * from "./fetch";
|
||||
export * from "./find";
|
||||
export * from "./gemini-image";
|
||||
export * from "./gh";
|
||||
@@ -219,7 +217,6 @@ export const BUILTIN_TOOLS: Record<string, ToolFactory> = {
|
||||
cancel_job: CancelJobTool.createIf,
|
||||
await: AwaitTool.createIf,
|
||||
todo_write: s => new TodoWriteTool(s),
|
||||
fetch: s => new FetchTool(s),
|
||||
web_search: s => new SearchTool(s),
|
||||
search_tool_bm25: SearchToolBm25Tool.createIf,
|
||||
write: s => new WriteTool(s),
|
||||
@@ -358,7 +355,6 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
if (name === "render_mermaid") return session.settings.get("renderMermaid.enabled");
|
||||
if (name === "notebook") return session.settings.get("notebook.enabled");
|
||||
if (name === "inspect_image") return session.settings.get("inspect_image.enabled");
|
||||
if (name === "fetch") return session.settings.get("fetch.enabled");
|
||||
if (name === "web_search") return session.settings.get("web_search.enabled");
|
||||
if (name === "search_tool_bm25") return session.settings.get("mcp.discoveryMode");
|
||||
if (name === "lsp") return session.settings.get("lsp.enabled");
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import * as url from "node:url";
|
||||
|
||||
const UNICODE_SPACES = /[\u00A0\u2000-\u200A\u202F\u205F\u3000]/g;
|
||||
const NARROW_NO_BREAK_SPACE = "\u202F";
|
||||
@@ -82,6 +83,16 @@ function normalizeAtPrefix(filePath: string): string {
|
||||
return filePath;
|
||||
}
|
||||
|
||||
function stripFileUrl(filePath: string): string {
|
||||
if (!filePath.toLowerCase().startsWith("file://")) return filePath;
|
||||
|
||||
try {
|
||||
return url.fileURLToPath(filePath);
|
||||
} catch {
|
||||
return filePath;
|
||||
}
|
||||
}
|
||||
|
||||
export function expandTilde(filePath: string, home?: string): string {
|
||||
const h = home ?? os.homedir();
|
||||
if (filePath === "~") return h;
|
||||
@@ -95,7 +106,7 @@ export function expandTilde(filePath: string, home?: string): string {
|
||||
}
|
||||
|
||||
export function expandPath(filePath: string): string {
|
||||
const normalized = normalizeUnicodeSpaces(normalizeAtPrefix(filePath));
|
||||
const normalized = stripFileUrl(normalizeUnicodeSpaces(normalizeAtPrefix(filePath)));
|
||||
return expandTilde(normalized);
|
||||
}
|
||||
|
||||
|
||||
@@ -35,9 +35,17 @@ import {
|
||||
import { convertFileWithMarkit } from "../utils/markit";
|
||||
import { detectSupportedImageMimeTypeFromFile } from "../utils/mime";
|
||||
import { type ArchiveReader, openArchive, parseArchivePathCandidates } from "./archive-reader";
|
||||
import {
|
||||
executeReadUrl,
|
||||
isReadableUrlPath,
|
||||
loadReadUrlCacheEntry,
|
||||
type ReadUrlToolDetails,
|
||||
renderReadUrlCall,
|
||||
renderReadUrlResult,
|
||||
} from "./fetch";
|
||||
import { applyListLimit } from "./list-limit";
|
||||
import { formatFullOutputReference, formatStyledTruncationWarning, type OutputMeta } from "./output-meta";
|
||||
import { resolveReadPath } from "./path-utils";
|
||||
import { expandPath, resolveReadPath } from "./path-utils";
|
||||
import { formatAge, formatBytes, shortenPath, wrapBrackets } from "./render-utils";
|
||||
import { ToolAbortError, ToolError, throwIfAborted } from "./tool-errors";
|
||||
import { toolResult } from "./tool-result";
|
||||
@@ -345,18 +353,26 @@ function prependSuffixResolutionNotice(text: string, suffixResolution?: { from:
|
||||
}
|
||||
|
||||
const readSchema = Type.Object({
|
||||
path: Type.String({ description: "Path to the file to read (relative or absolute)" }),
|
||||
offset: Type.Optional(Type.Number({ description: "Line number to start reading from (1-indexed)" })),
|
||||
limit: Type.Optional(Type.Number({ description: "Maximum number of lines to read" })),
|
||||
path: Type.String({ description: "Path or URL to read" }),
|
||||
offset: Type.Optional(Type.Number({ description: "Line number to start from (1-indexed)" })),
|
||||
limit: Type.Optional(Type.Number({ description: "Maximum number of lines" })),
|
||||
timeout: Type.Optional(Type.Number({ description: "Timeout in seconds (default: 20)" })),
|
||||
raw: Type.Optional(Type.Boolean({ description: "If set, returns raw content without transformations" })),
|
||||
});
|
||||
|
||||
export type ReadToolInput = Static<typeof readSchema>;
|
||||
|
||||
export interface ReadToolDetails {
|
||||
kind?: "file" | "url";
|
||||
truncation?: TruncationResult;
|
||||
isDirectory?: boolean;
|
||||
resolvedPath?: string;
|
||||
suffixResolution?: { from: string; to: string };
|
||||
url?: string;
|
||||
finalUrl?: string;
|
||||
contentType?: string;
|
||||
method?: string;
|
||||
notes?: string[];
|
||||
meta?: OutputMeta;
|
||||
}
|
||||
|
||||
@@ -451,6 +467,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
options: {
|
||||
details?: ReadToolDetails;
|
||||
sourcePath?: string;
|
||||
sourceUrl?: string;
|
||||
sourceInternal?: string;
|
||||
entityLabel: string;
|
||||
},
|
||||
@@ -466,6 +483,9 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
if (options.sourcePath) {
|
||||
resultBuilder.sourcePath(options.sourcePath);
|
||||
}
|
||||
if (options.sourceUrl) {
|
||||
resultBuilder.sourceUrl(options.sourceUrl);
|
||||
}
|
||||
if (options.sourceInternal) {
|
||||
resultBuilder.sourceInternal(options.sourceInternal);
|
||||
}
|
||||
@@ -651,9 +671,11 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
_onUpdate?: AgentToolUpdateCallback<ReadToolDetails>,
|
||||
_toolContext?: AgentToolContext,
|
||||
): Promise<AgentToolResult<ReadToolDetails>> {
|
||||
const { path: readPath, offset, limit } = params;
|
||||
|
||||
let { path: readPath, offset, limit, timeout, raw } = params;
|
||||
const displayMode = resolveFileDisplayMode(this.session);
|
||||
if (readPath.startsWith("file://")) {
|
||||
readPath = expandPath(readPath);
|
||||
}
|
||||
|
||||
// Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://)
|
||||
const internalRouter = this.session.internalRouter;
|
||||
@@ -661,6 +683,24 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
return this.#handleInternalUrl(readPath, offset, limit);
|
||||
}
|
||||
|
||||
if (isReadableUrlPath(readPath)) {
|
||||
if (!this.session.settings.get("fetch.enabled")) {
|
||||
throw new ToolError("URL reads are disabled by settings.");
|
||||
}
|
||||
if (offset !== undefined || limit !== undefined) {
|
||||
const cached = await loadReadUrlCacheEntry(this.session, { path: readPath, timeout, raw }, signal, {
|
||||
ensureArtifact: true,
|
||||
preferCached: true,
|
||||
});
|
||||
return this.#buildInMemoryTextResult(cached.output, offset, limit, {
|
||||
details: { ...cached.details },
|
||||
sourceUrl: cached.details.finalUrl,
|
||||
entityLabel: "URL output",
|
||||
});
|
||||
}
|
||||
return executeReadUrl(this.session, { path: readPath, timeout, raw }, signal);
|
||||
}
|
||||
|
||||
const archivePath = await this.#resolveArchiveReadPath(readPath, signal);
|
||||
if (archivePath) {
|
||||
return this.#readArchive(readPath, offset, limit, archivePath, signal);
|
||||
@@ -1128,10 +1168,16 @@ interface ReadRenderArgs {
|
||||
file_path?: string;
|
||||
offset?: number;
|
||||
limit?: number;
|
||||
timeout?: number;
|
||||
raw?: boolean;
|
||||
}
|
||||
|
||||
export const readToolRenderer = {
|
||||
renderCall(args: ReadRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {
|
||||
if (isReadableUrlPath(args.file_path || args.path || "")) {
|
||||
return renderReadUrlCall(args, _options, uiTheme);
|
||||
}
|
||||
|
||||
const rawPath = args.file_path || args.path || "";
|
||||
const filePath = shortenPath(rawPath);
|
||||
const offset = args.offset;
|
||||
@@ -1154,6 +1200,15 @@ export const readToolRenderer = {
|
||||
uiTheme: Theme,
|
||||
args?: ReadRenderArgs,
|
||||
): Component {
|
||||
const urlDetails = result.details as ReadUrlToolDetails | undefined;
|
||||
if (urlDetails?.kind === "url" || isReadableUrlPath(args?.file_path || args?.path || "")) {
|
||||
return renderReadUrlResult(
|
||||
result as { content: Array<{ type: string; text?: string }>; details?: ReadUrlToolDetails },
|
||||
_options,
|
||||
uiTheme,
|
||||
);
|
||||
}
|
||||
|
||||
const details = result.details;
|
||||
const contentText = result.content?.find(c => c.type === "text")?.text ?? "";
|
||||
const imageContent = result.content?.find(c => c.type === "image");
|
||||
|
||||
@@ -15,7 +15,6 @@ import { astEditToolRenderer } from "./ast-edit";
|
||||
import { astGrepToolRenderer } from "./ast-grep";
|
||||
import { bashToolRenderer } from "./bash";
|
||||
import { calculatorToolRenderer } from "./calculator";
|
||||
import { fetchToolRenderer } from "./fetch";
|
||||
import { findToolRenderer } from "./find";
|
||||
import { ghRunWatchToolRenderer } from "./gh-renderer";
|
||||
import { grepToolRenderer } from "./grep";
|
||||
@@ -61,7 +60,6 @@ export const toolRenderers: Record<string, ToolRenderer> = {
|
||||
ssh: sshToolRenderer as ToolRenderer,
|
||||
task: taskToolRenderer as ToolRenderer,
|
||||
todo_write: todoWriteToolRenderer as ToolRenderer,
|
||||
fetch: fetchToolRenderer as ToolRenderer,
|
||||
gh_run_watch: ghRunWatchToolRenderer as ToolRenderer,
|
||||
web_search: webSearchToolRenderer as ToolRenderer,
|
||||
write: writeToolRenderer as ToolRenderer,
|
||||
|
||||
@@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import * as url from "node:url";
|
||||
import * as zlib from "node:zlib";
|
||||
import type { AgentToolContext } from "@oh-my-pi/pi-agent-core";
|
||||
import { DEFAULT_BASH_INTERCEPTOR_RULES, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
@@ -301,6 +302,16 @@ describe("Coding Agent Tools", () => {
|
||||
await expect(readTool.execute("test-call-2", { path: testFile })).rejects.toThrow(/ENOENT|not found/i);
|
||||
});
|
||||
|
||||
it("should read local files passed as file:// URLs", async () => {
|
||||
const testFile = path.join(testDir, "file-url.txt");
|
||||
fs.writeFileSync(testFile, "Hello from file URL");
|
||||
|
||||
const result = await readTool.execute("test-call-file-url", { path: url.pathToFileURL(testFile).href });
|
||||
const output = getTextOutput(result);
|
||||
|
||||
expect(output).toContain("Hello from file URL");
|
||||
});
|
||||
|
||||
it("should truncate files exceeding line limit", async () => {
|
||||
const testFile = path.join(testDir, "large.txt");
|
||||
const lines = Array.from({ length: 3500 }, (_, i) => `Line ${i + 1}`);
|
||||
|
||||
@@ -4,7 +4,7 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { type SettingPath, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { FetchTool } from "@oh-my-pi/pi-coding-agent/tools/fetch";
|
||||
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import * as imageResize from "@oh-my-pi/pi-coding-agent/utils/image-resize";
|
||||
import * as toolsManager from "@oh-my-pi/pi-coding-agent/utils/tools-manager";
|
||||
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
@@ -21,7 +21,7 @@ const withMissingSystemPython = () => {
|
||||
};
|
||||
};
|
||||
|
||||
describe("fetch tool", () => {
|
||||
describe("read tool URL handling", () => {
|
||||
let testDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
@@ -37,11 +37,21 @@ describe("fetch tool", () => {
|
||||
|
||||
const createSession = (overrides: Partial<Record<SettingPath, unknown>> = {}): ToolSession => {
|
||||
const sessionFile = path.join(testDir, "session.jsonl");
|
||||
const artifactsDir = sessionFile.slice(0, -6);
|
||||
let nextArtifactId = 0;
|
||||
return {
|
||||
cwd: testDir,
|
||||
hasUI: false,
|
||||
getSessionFile: () => sessionFile,
|
||||
getArtifactsDir: () => artifactsDir,
|
||||
getSessionSpawns: () => null,
|
||||
allocateOutputArtifact: async toolType => {
|
||||
const id = String(nextArtifactId++);
|
||||
return {
|
||||
id,
|
||||
path: path.join(artifactsDir, `${id}.${toolType}.log`),
|
||||
};
|
||||
},
|
||||
settings: Settings.isolated({
|
||||
"fetch.enabled": true,
|
||||
...overrides,
|
||||
@@ -51,7 +61,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("returns an image content block when fetching image URLs", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const imageBytes = new Uint8Array([137, 80, 78, 71]);
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
@@ -82,7 +92,7 @@ describe("fetch tool", () => {
|
||||
},
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image", { url: "https://example.com/image.png" });
|
||||
const result = await tool.execute("fetch-image", { path: "https://example.com/image.png" });
|
||||
const imageBlock = result.content.find(
|
||||
(content): content is { type: "image"; data: string; mimeType: string } => content.type === "image",
|
||||
);
|
||||
@@ -95,7 +105,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("resizes fetched images before emitting image content blocks", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const resizeSpy = vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
|
||||
buffer: new Uint8Array([1, 2, 3]),
|
||||
mimeType: "image/jpeg",
|
||||
@@ -125,7 +135,7 @@ describe("fetch tool", () => {
|
||||
error: "markit unavailable",
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-resized", { url: "https://example.com/image.png" });
|
||||
const result = await tool.execute("fetch-image-resized", { path: "https://example.com/image.png" });
|
||||
const imageBlock = result.content.find(
|
||||
(content): content is { type: "image"; data: string; mimeType: string } => content.type === "image",
|
||||
);
|
||||
@@ -141,7 +151,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("keeps markit extracted text for image responses", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const extractedText = "Converted image text content that is definitely longer than fifty characters.";
|
||||
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
|
||||
buffer: new Uint8Array([1, 2, 3]),
|
||||
@@ -171,7 +181,7 @@ describe("fetch tool", () => {
|
||||
content: extractedText,
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-with-ocr", { url: "https://example.com/image.png" });
|
||||
const result = await tool.execute("fetch-image-with-ocr", { path: "https://example.com/image.png" });
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
const imageBlock = result.content.find(
|
||||
(content): content is { type: "image"; data: string; mimeType: string } => content.type === "image",
|
||||
@@ -185,7 +195,7 @@ describe("fetch tool", () => {
|
||||
});
|
||||
it("falls back to text-only output for unsupported image MIME types", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const fetchBinarySpy = vi.spyOn(scraperUtils, "fetchBinary");
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
@@ -195,7 +205,7 @@ describe("fetch tool", () => {
|
||||
content: "<svg></svg>",
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-unsupported", { url: "https://example.com/image.svg" });
|
||||
const result = await tool.execute("fetch-image-unsupported", { path: "https://example.com/image.svg" });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -208,7 +218,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("uses binary conversion fallback for unsupported image MIME when extension is convertible", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const convertedText = "Converted image text from markit fallback with sufficient length to pass threshold.";
|
||||
const fetchBinarySpy = vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
|
||||
ok: true,
|
||||
@@ -226,7 +236,7 @@ describe("fetch tool", () => {
|
||||
content: "\u0000\u0001garbage",
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-jpg-fallback", { url: "https://example.com/image.jpg" });
|
||||
const result = await tool.execute("fetch-image-jpg-fallback", { path: "https://example.com/image.jpg" });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -241,7 +251,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("does not treat text/html at .png paths as inline images", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({ ok: false, error: "not an image" });
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
@@ -251,7 +261,7 @@ describe("fetch tool", () => {
|
||||
content: "<html><body>not really an image</body></html>",
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-html-png-path", { url: "https://example.com/foo.png", raw: true });
|
||||
const result = await tool.execute("fetch-html-png-path", { path: "https://example.com/foo.png", raw: true });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -263,7 +273,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("falls back to textual output when inline image refetch fails", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit");
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
@@ -276,7 +286,7 @@ describe("fetch tool", () => {
|
||||
.spyOn(scraperUtils, "fetchBinary")
|
||||
.mockResolvedValue({ ok: false, error: "upstream blocked" });
|
||||
|
||||
const result = await tool.execute("fetch-image-refetch-failed", { url: "https://example.com/transient.png" });
|
||||
const result = await tool.execute("fetch-image-refetch-failed", { path: "https://example.com/transient.png" });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -289,7 +299,7 @@ describe("fetch tool", () => {
|
||||
});
|
||||
it("falls back to text-only output when image payload bytes are invalid", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
@@ -319,7 +329,7 @@ describe("fetch tool", () => {
|
||||
},
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-broken-image", { url: "https://example.com/broken.png" });
|
||||
const result = await tool.execute("fetch-broken-image", { path: "https://example.com/broken.png" });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -330,7 +340,7 @@ describe("fetch tool", () => {
|
||||
});
|
||||
it("prefers rendered page content over site-wide llms.txt for deep pages", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const pageUrl = "https://bun.com/reference/bun/UnixSocketOptions";
|
||||
const pageHtml = "<html><body><main><h1>UnixSocketOptions</h1><p>Page-specific docs.</p></main></body></html>";
|
||||
const renderedMarkdown = `# UnixSocketOptions\n\n${"Page-specific API docs. ".repeat(8)}`;
|
||||
@@ -378,7 +388,7 @@ describe("fetch tool", () => {
|
||||
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue(undefined);
|
||||
vi.spyOn(natives, "htmlToMarkdown").mockResolvedValue(renderedMarkdown);
|
||||
|
||||
const result = await tool.execute("fetch-deep-page", { url: pageUrl });
|
||||
const result = await tool.execute("fetch-deep-page", { path: pageUrl });
|
||||
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -392,7 +402,7 @@ describe("fetch tool", () => {
|
||||
|
||||
it("uses section-scoped llms.txt fallback without requesting the site-wide file", async () => {
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const pageUrl = "https://example.com/docs/reference/widget";
|
||||
const pageHtml = "<html><body><nav>Docs</nav><main><h1>Widget</h1></main></body></html>";
|
||||
const lowQualityRender = `${"Please enable JavaScript to view this page.\n".repeat(6)}${"navigation\n".repeat(4)}`;
|
||||
@@ -457,7 +467,7 @@ describe("fetch tool", () => {
|
||||
using _hook = hookFetch(() => new Response("blocked", { status: 500, statusText: "Blocked" }));
|
||||
vi.spyOn(toolsManager, "ensureTool").mockResolvedValue("/usr/bin/trafilatura");
|
||||
|
||||
const result = await tool.execute("fetch-section-llms", { url: pageUrl });
|
||||
const result = await tool.execute("fetch-section-llms", { path: pageUrl });
|
||||
const requestedUrls = loadPageSpy.mock.calls.map(([requestedUrl]) => requestedUrl);
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
@@ -473,7 +483,7 @@ describe("fetch tool", () => {
|
||||
it("prefers Parallel extract before other HTML renderers when configured", async () => {
|
||||
process.env.PARALLEL_API_KEY = "test-parallel-key";
|
||||
const session = createSession();
|
||||
const tool = new FetchTool(session);
|
||||
const tool = new ReadTool(session);
|
||||
const pageUrl = "https://example.com/parallel-page";
|
||||
const pageHtml = "<html><body><main><h1>Parallel Page</h1></main></body></html>";
|
||||
const ensureToolSpy = vi.spyOn(toolsManager, "ensureTool");
|
||||
@@ -536,7 +546,7 @@ describe("fetch tool", () => {
|
||||
return new Response("blocked", { status: 500, statusText: "Blocked" });
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-parallel-html", { url: pageUrl });
|
||||
const result = await tool.execute("fetch-parallel-html", { path: pageUrl });
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
expect(result.details?.method).toBe("parallel");
|
||||
@@ -545,4 +555,63 @@ describe("fetch tool", () => {
|
||||
expect(ensureToolSpy).not.toHaveBeenCalled();
|
||||
expect(htmlToMarkdownSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("reuses cached output for repeated plain URL reads", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
const pageUrl = "https://example.com/repeated-read-cache";
|
||||
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: pageUrl,
|
||||
content: "Cached line 1\nCached line 2",
|
||||
});
|
||||
|
||||
const firstResult = await tool.execute("fetch-cache-first", { path: pageUrl });
|
||||
const secondResult = await tool.execute("fetch-cache-second", { path: pageUrl });
|
||||
const firstText = firstResult.content.find(content => content.type === "text");
|
||||
const secondText = secondResult.content.find(content => content.type === "text");
|
||||
|
||||
expect(firstText?.type).toBe("text");
|
||||
expect(firstText?.text).toContain("Cached line 1");
|
||||
expect(secondText?.type).toBe("text");
|
||||
expect(secondText?.text).toContain("Cached line 1");
|
||||
expect(loadPageSpy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("supports offset and limit for URL reads using cached output", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
const pageUrl = "https://example.com/offset-test";
|
||||
const loadPageSpy = vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: pageUrl,
|
||||
content: "Line 1\nLine 2\nLine 3\nLine 4",
|
||||
});
|
||||
|
||||
const firstResult = await tool.execute("fetch-offset-prime", { path: pageUrl });
|
||||
const firstText = firstResult.content.find(content => content.type === "text");
|
||||
expect(firstText?.type).toBe("text");
|
||||
expect(firstText?.text).toContain("Line 1");
|
||||
expect(loadPageSpy).toHaveBeenCalledTimes(1);
|
||||
|
||||
loadPageSpy.mockClear();
|
||||
loadPageSpy.mockRejectedValue(new Error("network should not be hit"));
|
||||
|
||||
const pagedResult = await tool.execute("fetch-offset-page", {
|
||||
path: pageUrl,
|
||||
offset: 7,
|
||||
limit: 2,
|
||||
});
|
||||
const pagedText = pagedResult.content.find(content => content.type === "text");
|
||||
expect(pagedText?.type).toBe("text");
|
||||
expect(pagedText?.text).toContain("Line 1");
|
||||
expect(pagedText?.text).toContain("Line 2");
|
||||
expect(pagedText?.text).not.toContain("Line 3");
|
||||
expect(loadPageSpy).not.toHaveBeenCalled();
|
||||
expect(fs.readdirSync(path.join(testDir, "session")).some(file => file.endsWith(".read.log"))).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -64,9 +64,9 @@ describe("createTools", () => {
|
||||
expect(names).toContain("notebook");
|
||||
expect(names).toContain("task");
|
||||
expect(names).toContain("todo_write");
|
||||
expect(names).toContain("fetch");
|
||||
expect(names).toContain("web_search");
|
||||
expect(names).toContain("exit_plan_mode");
|
||||
expect(names).not.toContain("fetch");
|
||||
});
|
||||
|
||||
it("includes bash and python when python mode is both", async () => {
|
||||
|
||||
Reference in New Issue
Block a user