feat(coding-agent/tools): added binary routing for notebook, sqlite, and archive payloads
- Added archive format sniffing to identify ZIP, TAR, and TAR.GZ from file bytes. - Added MIME/extension and header-based routing for notebook, sqlite, and archive payloads. - Added archive entry rendering with slash-terminated dirs and size suffixes. - Added tests for archive, sqlite, notebook, and fallback binary dispatch scenarios.
This commit is contained in:
@@ -1,5 +1,9 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { inflateSync, strFromU8 } from "fflate";
|
||||
|
||||
import { formatBytes } from "./render-utils";
|
||||
import { ToolError } from "./tool-errors";
|
||||
|
||||
export type ArchiveFormat = "zip" | "tar" | "tar.gz";
|
||||
@@ -123,11 +127,21 @@ function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] {
|
||||
return entries.map(entry => {
|
||||
if (entry.isDirectory) return `${entry.name}/`;
|
||||
|
||||
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
|
||||
return `${entry.name}${sizeSuffix}`;
|
||||
});
|
||||
}
|
||||
|
||||
const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50;
|
||||
const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50;
|
||||
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
|
||||
const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50;
|
||||
const ZIP_EOCD_SIGNATURE = 0x06054b50;
|
||||
const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50;
|
||||
const ZIP_EOCD_MIN_LENGTH = 22;
|
||||
const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff;
|
||||
const ZIP64_EOCD_LOCATOR_LENGTH = 20;
|
||||
@@ -167,6 +181,37 @@ function readUInt32LE(bytes: Uint8Array, offset: number): number {
|
||||
return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0;
|
||||
}
|
||||
|
||||
function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean {
|
||||
if (bytes.byteLength < offset + value.length) return false;
|
||||
for (let index = 0; index < value.length; index++) {
|
||||
if (bytes[offset + index] !== value.charCodeAt(index)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined {
|
||||
if (bytes.byteLength >= 4) {
|
||||
const signature = readUInt32LE(bytes, 0);
|
||||
if (
|
||||
signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE ||
|
||||
signature === ZIP_EOCD_SIGNATURE ||
|
||||
signature === ZIP_DATA_DESCRIPTOR_SIGNATURE
|
||||
) {
|
||||
return "zip";
|
||||
}
|
||||
}
|
||||
|
||||
if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) {
|
||||
return "tar.gz";
|
||||
}
|
||||
|
||||
if (bytesMatchAscii(bytes, 257, "ustar")) {
|
||||
return "tar";
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number {
|
||||
const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE;
|
||||
if (!Number.isSafeInteger(value)) {
|
||||
@@ -627,3 +672,22 @@ export async function openArchive(filePath: string): Promise<ArchiveReader> {
|
||||
format === "zip" ? await readZipEntries(filePath) : await readTarEntries(await Bun.file(filePath).bytes());
|
||||
return new ArchiveReader(format, entries);
|
||||
}
|
||||
|
||||
export async function listArchiveRoot(
|
||||
bytes: Uint8Array,
|
||||
format: ArchiveFormat,
|
||||
opts: { limit?: number } = {},
|
||||
): Promise<string> {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-archive-"));
|
||||
const tempPath = path.join(tempDir, `payload.${format}`);
|
||||
try {
|
||||
await Bun.write(tempPath, bytes);
|
||||
const archive = await openArchive(tempPath);
|
||||
const entries = archive.listDirectory("");
|
||||
const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries;
|
||||
const lines = formatArchiveEntryLines(limitedEntries);
|
||||
return lines.length > 0 ? lines.join("\n") : "(empty archive directory)";
|
||||
} finally {
|
||||
await fs.rm(tempDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
import { Database } from "bun:sqlite";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
|
||||
@@ -8,6 +10,7 @@ import { $which, ptree, truncate } from "@oh-my-pi/pi-utils";
|
||||
import { parseHTML } from "linkedom";
|
||||
import { LRUCache } from "lru-cache/raw";
|
||||
import type { Settings } from "../config/settings";
|
||||
import { readEditableNotebookText } from "../edit/notebook";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
import { type Theme, theme } from "../modes/theme/theme";
|
||||
import type { ToolSession } from "../sdk";
|
||||
@@ -22,10 +25,12 @@ import { specialHandlers } from "../web/scrapers";
|
||||
import type { RenderResult } from "../web/scrapers/types";
|
||||
import { finalizeOutput, loadPage, looksLikeHtml, MAX_OUTPUT_CHARS } from "../web/scrapers/types";
|
||||
import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils";
|
||||
import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "./archive-reader";
|
||||
import { applyListLimit } from "./list-limit";
|
||||
import { formatStyledArtifactReference, type OutputMeta } from "./output-meta";
|
||||
import { type LineRange, parseLineRanges } from "./path-utils";
|
||||
import { formatExpandHint, getDomain, replaceTabs } from "./render-utils";
|
||||
import { formatBytes, formatExpandHint, getDomain, replaceTabs } from "./render-utils";
|
||||
import { listTables, looksLikeSqlite, renderTableList } from "./sqlite-reader";
|
||||
import { ToolAbortError, ToolError } from "./tool-errors";
|
||||
import { toolResult } from "./tool-result";
|
||||
import { clampTimeout } from "./tool-timeouts";
|
||||
@@ -46,8 +51,6 @@ const CONVERTIBLE_MIMES = new Set([
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"application/x-ipynb+json",
|
||||
"application/zip",
|
||||
"image/png",
|
||||
"image/jpeg",
|
||||
"image/gif",
|
||||
@@ -67,7 +70,6 @@ const CONVERTIBLE_EXTENSIONS = new Set([
|
||||
".xlsx",
|
||||
".rtf",
|
||||
".epub",
|
||||
".ipynb",
|
||||
".png",
|
||||
".jpg",
|
||||
".jpeg",
|
||||
@@ -78,6 +80,27 @@ const CONVERTIBLE_EXTENSIONS = new Set([
|
||||
".ogg",
|
||||
]);
|
||||
|
||||
const NOTEBOOK_MIMES = new Set(["application/x-ipynb+json"]);
|
||||
const NOTEBOOK_EXTENSIONS = new Set([".ipynb"]);
|
||||
|
||||
const SQLITE_MIMES = new Set([
|
||||
"application/vnd.sqlite3",
|
||||
"application/x-sqlite3",
|
||||
"application/sqlite3",
|
||||
"application/sqlite",
|
||||
]);
|
||||
const SQLITE_EXTENSIONS = new Set([".sqlite", ".sqlite3", ".db", ".db3"]);
|
||||
|
||||
const ARCHIVE_MIMES = new Set([
|
||||
"application/zip",
|
||||
"application/x-zip-compressed",
|
||||
"application/x-tar",
|
||||
"application/tar",
|
||||
"application/gzip",
|
||||
"application/x-gzip",
|
||||
]);
|
||||
const ARCHIVE_EXTENSIONS = new Set([".zip", ".tar", ".tar.gz", ".tgz", ".gz"]);
|
||||
|
||||
const IMAGE_MIME_BY_EXTENSION = new Map<string, string>([
|
||||
[".png", "image/png"],
|
||||
[".jpg", "image/jpeg"],
|
||||
@@ -261,6 +284,12 @@ function normalizeMime(contentType: string): string {
|
||||
return contentType.split(";")[0].trim().toLowerCase();
|
||||
}
|
||||
|
||||
function getFilenameExtensionHint(filename: string): string {
|
||||
const lower = filename.toLowerCase();
|
||||
if (lower.endsWith(".tar.gz")) return ".tar.gz";
|
||||
return path.extname(filename).toLowerCase();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get extension from URL or Content-Disposition
|
||||
*/
|
||||
@@ -269,7 +298,7 @@ function getExtensionHint(url: string, contentDisposition?: string): string {
|
||||
if (contentDisposition) {
|
||||
const match = contentDisposition.match(/filename[*]?=["']?([^"';\n]+)/i);
|
||||
if (match) {
|
||||
const ext = path.extname(match[1]).toLowerCase();
|
||||
const ext = getFilenameExtensionHint(match[1]);
|
||||
if (ext) return ext;
|
||||
}
|
||||
}
|
||||
@@ -277,7 +306,7 @@ function getExtensionHint(url: string, contentDisposition?: string): string {
|
||||
// Fall back to URL path
|
||||
try {
|
||||
const pathname = new URL(url).pathname;
|
||||
const ext = path.extname(pathname).toLowerCase();
|
||||
const ext = getFilenameExtensionHint(pathname);
|
||||
if (ext) return ext;
|
||||
} catch {}
|
||||
|
||||
@@ -738,6 +767,254 @@ type FetchRenderResult = RenderResult & {
|
||||
image?: FetchImagePayload;
|
||||
};
|
||||
|
||||
const BINARY_SAMPLE_CHARS = 4096;
|
||||
const URL_ARCHIVE_LIST_LIMIT = 500;
|
||||
const URL_SQLITE_LIST_LIMIT = 500;
|
||||
|
||||
function sampleLooksBinary(text: string): boolean {
|
||||
const limit = Math.min(text.length, BINARY_SAMPLE_CHARS);
|
||||
if (limit === 0) return false;
|
||||
|
||||
let replacementCount = 0;
|
||||
for (let index = 0; index < limit; index++) {
|
||||
const code = text.charCodeAt(index);
|
||||
if (code === 0) return true;
|
||||
if (code === 0xfffd) replacementCount++;
|
||||
}
|
||||
|
||||
return replacementCount >= 3 && replacementCount / limit > 0.01;
|
||||
}
|
||||
|
||||
function isNotebookHint(mime: string, extensionHint: string): boolean {
|
||||
return NOTEBOOK_MIMES.has(mime) || NOTEBOOK_EXTENSIONS.has(extensionHint);
|
||||
}
|
||||
|
||||
function isSqliteHint(mime: string, extensionHint: string): boolean {
|
||||
return SQLITE_MIMES.has(mime) || SQLITE_EXTENSIONS.has(extensionHint);
|
||||
}
|
||||
|
||||
function isArchiveHint(mime: string, extensionHint: string): boolean {
|
||||
return ARCHIVE_MIMES.has(mime) || ARCHIVE_EXTENSIONS.has(extensionHint);
|
||||
}
|
||||
|
||||
function getArchiveFormatHint(mime: string, extensionHint: string): ArchiveFormat | undefined {
|
||||
if (extensionHint === ".zip" || mime === "application/zip" || mime === "application/x-zip-compressed") {
|
||||
return "zip";
|
||||
}
|
||||
if (extensionHint === ".tar" || mime === "application/x-tar" || mime === "application/tar") {
|
||||
return "tar";
|
||||
}
|
||||
if (
|
||||
extensionHint === ".tar.gz" ||
|
||||
extensionHint === ".tgz" ||
|
||||
extensionHint === ".gz" ||
|
||||
mime === "application/gzip" ||
|
||||
mime === "application/x-gzip"
|
||||
) {
|
||||
return "tar.gz";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function formatErrorMessage(error: unknown): string {
|
||||
return error instanceof Error ? error.message : String(error);
|
||||
}
|
||||
|
||||
function binaryContentType(mime: string): string {
|
||||
return mime || "application/octet-stream";
|
||||
}
|
||||
|
||||
function buildBinaryNotice(finalUrl: string, mime: string, byteLength?: number): string {
|
||||
const size = byteLength === undefined ? "unknown size" : formatBytes(byteLength);
|
||||
return `[Binary content: ${binaryContentType(mime)}, ${size}] ${finalUrl}`;
|
||||
}
|
||||
|
||||
function buildBinaryPayloadResult(
|
||||
url: string,
|
||||
finalUrl: string,
|
||||
mime: string,
|
||||
method: string,
|
||||
content: string,
|
||||
fetchedAt: string,
|
||||
notes: string[],
|
||||
): FetchRenderResult {
|
||||
const output = finalizeOutput(content);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: binaryContentType(mime),
|
||||
method,
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
async function withTempBinaryFile<T>(
|
||||
prefix: string,
|
||||
extension: string,
|
||||
bytes: Uint8Array,
|
||||
readTempFile: (tempPath: string) => Promise<T>,
|
||||
): Promise<T> {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), prefix));
|
||||
const tempPath = path.join(tempDir, `payload${extension}`);
|
||||
try {
|
||||
await Bun.write(tempPath, bytes);
|
||||
return await readTempFile(tempPath);
|
||||
} finally {
|
||||
await fs.rm(tempDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
async function renderNotebookPayload(bytes: Uint8Array, displayUrl: string): Promise<string> {
|
||||
return withTempBinaryFile("omp-url-notebook-", ".ipynb", bytes, tempPath =>
|
||||
readEditableNotebookText(tempPath, displayUrl),
|
||||
);
|
||||
}
|
||||
|
||||
async function renderSqlitePayload(bytes: Uint8Array): Promise<string> {
|
||||
return withTempBinaryFile("omp-url-sqlite-", ".sqlite", bytes, async tempPath => {
|
||||
let db: Database | null = null;
|
||||
try {
|
||||
db = new Database(tempPath, { readonly: true, strict: true });
|
||||
db.run("PRAGMA busy_timeout = 3000");
|
||||
const listLimit = applyListLimit(listTables(db), { limit: URL_SQLITE_LIST_LIMIT });
|
||||
return renderTableList(listLimit.items);
|
||||
} finally {
|
||||
db?.close();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
async function tryRenderBinaryPayload(
|
||||
url: string,
|
||||
finalUrl: string,
|
||||
mime: string,
|
||||
extHint: string,
|
||||
rawContent: string,
|
||||
timeout: number,
|
||||
signal: AbortSignal | undefined,
|
||||
fetchedAt: string,
|
||||
notes: readonly string[],
|
||||
): Promise<FetchRenderResult | null> {
|
||||
const hasNotebookHint = isNotebookHint(mime, extHint);
|
||||
const hasSqliteHint = isSqliteHint(mime, extHint);
|
||||
const hasArchiveHint = isArchiveHint(mime, extHint);
|
||||
const rawLooksBinary = sampleLooksBinary(rawContent);
|
||||
if (!hasNotebookHint && !hasSqliteHint && !hasArchiveHint && !rawLooksBinary) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const resultNotes = [...notes];
|
||||
const binary = await fetchBinary(finalUrl, timeout, signal);
|
||||
if (!binary.ok) {
|
||||
resultNotes.push(binary.error ? `Binary fetch failed: ${binary.error}` : "Binary fetch failed");
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"binary",
|
||||
buildBinaryNotice(finalUrl, mime),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
}
|
||||
|
||||
const binaryExtHint = getExtensionHint(finalUrl, binary.contentDisposition) || extHint;
|
||||
if (isNotebookHint(mime, binaryExtHint)) {
|
||||
try {
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"notebook",
|
||||
await renderNotebookPayload(binary.buffer, finalUrl),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
} catch (error) {
|
||||
resultNotes.push(`Notebook rendering failed: ${formatErrorMessage(error)}`);
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"binary",
|
||||
buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (isSqliteHint(mime, binaryExtHint) || looksLikeSqlite(binary.buffer)) {
|
||||
try {
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"sqlite",
|
||||
await renderSqlitePayload(binary.buffer),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
} catch (error) {
|
||||
resultNotes.push(`SQLite rendering failed: ${formatErrorMessage(error)}`);
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"binary",
|
||||
buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const hintedArchiveFormat = getArchiveFormatHint(mime, binaryExtHint);
|
||||
const shouldArchiveSniff = hintedArchiveFormat !== undefined || !isConvertible(mime, binaryExtHint);
|
||||
const archiveFormat = hintedArchiveFormat ?? (shouldArchiveSniff ? sniffArchiveFormat(binary.buffer) : undefined);
|
||||
if (archiveFormat) {
|
||||
try {
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"archive",
|
||||
await listArchiveRoot(binary.buffer, archiveFormat, { limit: URL_ARCHIVE_LIST_LIMIT }),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
} catch (error) {
|
||||
resultNotes.push(`Archive rendering failed: ${formatErrorMessage(error)}`);
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"binary",
|
||||
buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (rawLooksBinary) {
|
||||
return buildBinaryPayloadResult(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
"binary",
|
||||
buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength),
|
||||
fetchedAt,
|
||||
resultNotes,
|
||||
);
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Unified Special Handler Dispatch
|
||||
// =============================================================================
|
||||
@@ -984,6 +1261,19 @@ async function renderUrl(
|
||||
}
|
||||
}
|
||||
|
||||
const binaryPayloadResult = await tryRenderBinaryPayload(
|
||||
url,
|
||||
finalUrl,
|
||||
mime,
|
||||
extHint,
|
||||
rawContent,
|
||||
timeout,
|
||||
signal,
|
||||
fetchedAt,
|
||||
notes,
|
||||
);
|
||||
if (binaryPayloadResult) return binaryPayloadResult;
|
||||
|
||||
// Step 4: Handle non-HTML text content
|
||||
const isHtml = mime.includes("html") || mime.includes("xhtml");
|
||||
const isJson = mime.includes("json");
|
||||
@@ -992,7 +1282,7 @@ async function renderUrl(
|
||||
const isFeed = mime.includes("rss") || mime.includes("atom") || mime.includes("feed");
|
||||
|
||||
// Raw mode skips every text-shaping branch below (JSON pretty-print, feed-to-markdown,
|
||||
// HTML extraction) and returns the response body verbatim. The image/markit branches
|
||||
// HTML extraction) and returns the response body verbatim. Binary-oriented branches
|
||||
// above already ran because raw isn't useful for binary payloads.
|
||||
if (raw) {
|
||||
const output = finalizeOutput(rawContent);
|
||||
|
||||
@@ -34,7 +34,7 @@ import { resolveFileDisplayMode } from "../utils/file-display-mode";
|
||||
import { ImageInputTooLargeError, loadImageInput, MAX_IMAGE_INPUT_BYTES } from "../utils/image-loading";
|
||||
import { convertFileWithMarkit } from "../utils/markit";
|
||||
import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree";
|
||||
import { type ArchiveReader, openArchive, parseArchivePathCandidates } from "./archive-reader";
|
||||
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader";
|
||||
import {
|
||||
type ConflictEntry,
|
||||
type ConflictScope,
|
||||
@@ -1154,17 +1154,10 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
const limitedEntries = listLimit.items;
|
||||
const limitMeta = listLimit.meta;
|
||||
|
||||
const results: string[] = [];
|
||||
for (const entry of limitedEntries) {
|
||||
for (let index = 0; index < limitedEntries.length; index++) {
|
||||
throwIfAborted(signal);
|
||||
if (entry.isDirectory) {
|
||||
results.push(`${entry.name}/`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
|
||||
results.push(`${entry.name}${sizeSuffix}`);
|
||||
}
|
||||
const results = formatArchiveEntryLines(limitedEntries);
|
||||
|
||||
const output = results.length > 0 ? results.join("\n") : "(empty archive directory)";
|
||||
const text = prependSuffixResolutionNotice(output, details.suffixResolution);
|
||||
|
||||
@@ -5,6 +5,14 @@ import { ToolError } from "./tool-errors";
|
||||
const SQLITE_MAGIC = new Uint8Array([
|
||||
0x53, 0x51, 0x4c, 0x69, 0x74, 0x65, 0x20, 0x66, 0x6f, 0x72, 0x6d, 0x61, 0x74, 0x20, 0x33, 0x00,
|
||||
]);
|
||||
|
||||
export function looksLikeSqlite(bytes: Uint8Array): boolean {
|
||||
if (bytes.byteLength < SQLITE_MAGIC.byteLength) return false;
|
||||
for (const [index, byte] of SQLITE_MAGIC.entries()) {
|
||||
if (bytes[index] !== byte) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
const SQLITE_PATH_PATTERN = /\.(?:sqlite3?|db3?)(?=(?::|\?|$))/gi;
|
||||
const DEFAULT_QUERY_LIMIT = 20;
|
||||
const DEFAULT_SCHEMA_SAMPLE_LIMIT = 5;
|
||||
@@ -443,18 +451,7 @@ export function parseSqlitePathCandidates(filePath: string): SqlitePathCandidate
|
||||
|
||||
export async function isSqliteFile(absolutePath: string): Promise<boolean> {
|
||||
try {
|
||||
const bytes = await Bun.file(absolutePath).slice(0, SQLITE_MAGIC.byteLength).bytes();
|
||||
if (bytes.length !== SQLITE_MAGIC.byteLength) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (const [index, byte] of SQLITE_MAGIC.entries()) {
|
||||
if (bytes[index] !== byte) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
return looksLikeSqlite(await Bun.file(absolutePath).slice(0, SQLITE_MAGIC.byteLength).bytes());
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,217 @@
|
||||
import { Database } from "bun:sqlite";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils";
|
||||
import { Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import { zipSync } from "fflate";
|
||||
|
||||
function makeSession(testDir: string): ToolSession {
|
||||
const sessionFile = path.join(testDir, "session.jsonl");
|
||||
const artifactsDir = sessionFile.slice(0, -6);
|
||||
let nextArtifactId = 0;
|
||||
return {
|
||||
cwd: testDir,
|
||||
hasUI: false,
|
||||
getSessionFile: () => sessionFile,
|
||||
getArtifactsDir: () => artifactsDir,
|
||||
getSessionSpawns: () => null,
|
||||
allocateOutputArtifact: async toolType => {
|
||||
const id = String(nextArtifactId++);
|
||||
return { id, path: path.join(artifactsDir, `${id}.${toolType}.log`) };
|
||||
},
|
||||
settings: Settings.isolated({ "fetch.enabled": true }),
|
||||
};
|
||||
}
|
||||
|
||||
function stubUrlBytes(bytes: Uint8Array, contentType: string, contentDisposition?: string) {
|
||||
const decoded = Buffer.from(bytes).toString("utf-8");
|
||||
vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({
|
||||
ok: true,
|
||||
status: 200,
|
||||
finalUrl: requestedUrl,
|
||||
contentType,
|
||||
content: decoded,
|
||||
}));
|
||||
return vi.spyOn(scraperUtils, "fetchBinary").mockImplementation(async () => ({
|
||||
ok: true,
|
||||
buffer: bytes,
|
||||
contentDisposition,
|
||||
}));
|
||||
}
|
||||
|
||||
function stubUrlText(body: string, contentType: string) {
|
||||
vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({
|
||||
ok: true,
|
||||
status: 200,
|
||||
finalUrl: requestedUrl,
|
||||
contentType,
|
||||
content: body,
|
||||
}));
|
||||
return vi.spyOn(scraperUtils, "fetchBinary").mockImplementation(async () => {
|
||||
throw new Error("unexpected binary fetch");
|
||||
});
|
||||
}
|
||||
|
||||
function textOutput(result: { content: Array<TextContent | ImageContent> }): string {
|
||||
return result.content
|
||||
.filter((content): content is TextContent => content.type === "text")
|
||||
.map(content => content.text)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
async function createSqliteFixtureBytes(testDir: string): Promise<Uint8Array> {
|
||||
const dbPath = path.join(testDir, `url-fixture-${Snowflake.next()}.sqlite`);
|
||||
const db = new Database(dbPath);
|
||||
try {
|
||||
db.exec(`
|
||||
CREATE TABLE notes (id INTEGER PRIMARY KEY, title TEXT NOT NULL);
|
||||
INSERT INTO notes (title) VALUES ('alpha'), ('beta');
|
||||
`);
|
||||
} finally {
|
||||
db.close();
|
||||
}
|
||||
return Bun.file(dbPath).bytes();
|
||||
}
|
||||
|
||||
function createNotebookFixtureBytes(): Uint8Array {
|
||||
return Buffer.from(
|
||||
JSON.stringify({
|
||||
cells: [
|
||||
{ cell_type: "markdown", metadata: {}, source: ["# Remote notebook\n", "Rendered as editable cells"] },
|
||||
{ cell_type: "code", execution_count: null, metadata: {}, outputs: [], source: ["answer = 42\n"] },
|
||||
],
|
||||
metadata: {},
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
function uniqueUrl(name: string, extension: string): string {
|
||||
return `https://example.com/${name}-${Snowflake.next()}${extension}`;
|
||||
}
|
||||
|
||||
describe("read URL binary dispatch", () => {
|
||||
let testDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
testDir = path.join(os.tmpdir(), `fetch-binary-dispatch-${Snowflake.next()}`);
|
||||
fs.mkdirSync(testDir, { recursive: true });
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
fs.rmSync(testDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("lists a remote zip instead of dumping decoded bytes", async () => {
|
||||
const zipBytes = zipSync({
|
||||
"root.txt": Buffer.from("root file\n"),
|
||||
"nested/data.txt": Buffer.from("nested file\n"),
|
||||
});
|
||||
const url = uniqueUrl("archive", ".zip");
|
||||
stubUrlBytes(zipBytes, "application/octet-stream");
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-zip", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("archive");
|
||||
expect(text).toContain("Method: archive");
|
||||
expect(text).toContain("root.txt");
|
||||
expect(text).toContain("nested/");
|
||||
expect(text).not.toContain("PK\u0003\u0004");
|
||||
expect(text).not.toContain("�");
|
||||
});
|
||||
|
||||
it("returns a metadata notice when a hinted binary refetch fails", async () => {
|
||||
const url = uniqueUrl("oversized", ".zip");
|
||||
vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({
|
||||
ok: true,
|
||||
status: 200,
|
||||
finalUrl: requestedUrl,
|
||||
contentType: "application/octet-stream",
|
||||
content: "PK\u0003\u0004\u0000\u0001",
|
||||
}));
|
||||
vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
|
||||
ok: false,
|
||||
error: "content-length 52428801 exceeds 52428800",
|
||||
});
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-zip-too-large", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("binary");
|
||||
expect(result.details?.notes).toContain("Binary fetch failed: content-length 52428801 exceeds 52428800");
|
||||
expect(text).toContain("[Binary content: application/octet-stream");
|
||||
expect(text).not.toContain("PK\u0003\u0004");
|
||||
});
|
||||
|
||||
it("renders a remote sqlite database through the sqlite reader", async () => {
|
||||
const sqliteBytes = await createSqliteFixtureBytes(testDir);
|
||||
const url = uniqueUrl("data", ".db");
|
||||
stubUrlBytes(sqliteBytes, "application/octet-stream");
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-sqlite", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("sqlite");
|
||||
expect(text).toContain("Method: sqlite");
|
||||
expect(text).toContain("notes (2 rows)");
|
||||
});
|
||||
|
||||
it("renders a remote notebook as editable cells", async () => {
|
||||
const notebookBytes = createNotebookFixtureBytes();
|
||||
const url = uniqueUrl("notebook", ".ipynb");
|
||||
stubUrlBytes(notebookBytes, "application/octet-stream");
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-notebook", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("notebook");
|
||||
expect(text).toContain("Method: notebook");
|
||||
expect(text).toContain("# %% [markdown] cell:0");
|
||||
expect(text).toContain("# %% [code] cell:1");
|
||||
expect(text).toContain("answer = 42");
|
||||
});
|
||||
|
||||
it("returns a metadata notice for unrenderable binary bytes", async () => {
|
||||
const binaryBytes = new Uint8Array([0, 1, 2, 3, 255, 254, 253, 0, 7, 8]);
|
||||
const url = uniqueUrl("payload", ".bin");
|
||||
stubUrlBytes(binaryBytes, "application/octet-stream");
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-binary", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("binary");
|
||||
expect(text).toContain("Method: binary");
|
||||
expect(text).toContain("[Binary content: application/octet-stream");
|
||||
expect(text).not.toContain("\u0000");
|
||||
expect(text).not.toContain("�");
|
||||
});
|
||||
|
||||
it("leaves valid UTF-8 octet-stream payloads on the text path", async () => {
|
||||
const url = uniqueUrl("plain", ".txt");
|
||||
const fetchBinarySpy = stubUrlText("plain UTF-8 text\nsecond line", "application/octet-stream");
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("read-url-text-octet", { path: url });
|
||||
const text = textOutput(result);
|
||||
|
||||
expect(result.details?.method).toBe("raw");
|
||||
expect(text).toContain("plain UTF-8 text");
|
||||
expect(text).toContain("second line");
|
||||
expect(fetchBinarySpy).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user