From cbd7c201055518e0da4df30fc8d70edd12702bbc Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 7 Jun 2026 06:55:52 +0200 Subject: [PATCH] feat(coding-agent/tools): added binary routing for notebook, sqlite, and archive payloads - Added archive format sniffing to identify ZIP, TAR, and TAR.GZ from file bytes. - Added MIME/extension and header-based routing for notebook, sqlite, and archive payloads. - Added archive entry rendering with slash-terminated dirs and size suffixes. - Added tests for archive, sqlite, notebook, and fallback binary dispatch scenarios. --- .../coding-agent/src/tools/archive-reader.ts | 64 ++++ packages/coding-agent/src/tools/fetch.ts | 304 +++++++++++++++++- packages/coding-agent/src/tools/read.ts | 13 +- .../coding-agent/src/tools/sqlite-reader.ts | 21 +- .../test/tools/fetch-binary-dispatch.test.ts | 217 +++++++++++++ 5 files changed, 590 insertions(+), 29 deletions(-) create mode 100644 packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts diff --git a/packages/coding-agent/src/tools/archive-reader.ts b/packages/coding-agent/src/tools/archive-reader.ts index 11b09b0ad..cdcd8ae63 100644 --- a/packages/coding-agent/src/tools/archive-reader.ts +++ b/packages/coding-agent/src/tools/archive-reader.ts @@ -1,5 +1,9 @@ +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { inflateSync, strFromU8 } from "fflate"; +import { formatBytes } from "./render-utils"; import { ToolError } from "./tool-errors"; export type ArchiveFormat = "zip" | "tar" | "tar.gz"; @@ -123,11 +127,21 @@ function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined { return undefined; } +export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] { + return entries.map(entry => { + if (entry.isDirectory) return `${entry.name}/`; + + const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : ""; + return `${entry.name}${sizeSuffix}`; + }); +} + const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50; const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50; const ZIP64_EOCD_SIGNATURE = 0x06064b50; const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50; const ZIP_EOCD_SIGNATURE = 0x06054b50; +const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50; const ZIP_EOCD_MIN_LENGTH = 22; const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff; const ZIP64_EOCD_LOCATOR_LENGTH = 20; @@ -167,6 +181,37 @@ function readUInt32LE(bytes: Uint8Array, offset: number): number { return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0; } +function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean { + if (bytes.byteLength < offset + value.length) return false; + for (let index = 0; index < value.length; index++) { + if (bytes[offset + index] !== value.charCodeAt(index)) return false; + } + return true; +} + +export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined { + if (bytes.byteLength >= 4) { + const signature = readUInt32LE(bytes, 0); + if ( + signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE || + signature === ZIP_EOCD_SIGNATURE || + signature === ZIP_DATA_DESCRIPTOR_SIGNATURE + ) { + return "zip"; + } + } + + if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) { + return "tar.gz"; + } + + if (bytesMatchAscii(bytes, 257, "ustar")) { + return "tar"; + } + + return undefined; +} + function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number { const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE; if (!Number.isSafeInteger(value)) { @@ -627,3 +672,22 @@ export async function openArchive(filePath: string): Promise { format === "zip" ? await readZipEntries(filePath) : await readTarEntries(await Bun.file(filePath).bytes()); return new ArchiveReader(format, entries); } + +export async function listArchiveRoot( + bytes: Uint8Array, + format: ArchiveFormat, + opts: { limit?: number } = {}, +): Promise { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-archive-")); + const tempPath = path.join(tempDir, `payload.${format}`); + try { + await Bun.write(tempPath, bytes); + const archive = await openArchive(tempPath); + const entries = archive.listDirectory(""); + const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries; + const lines = formatArchiveEntryLines(limitedEntries); + return lines.length > 0 ? lines.join("\n") : "(empty archive directory)"; + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } +} diff --git a/packages/coding-agent/src/tools/fetch.ts b/packages/coding-agent/src/tools/fetch.ts index 4e2dcbecf..4d16ced81 100644 --- a/packages/coding-agent/src/tools/fetch.ts +++ b/packages/coding-agent/src/tools/fetch.ts @@ -1,4 +1,6 @@ +import { Database } from "bun:sqlite"; import * as fs from "node:fs/promises"; +import * as os from "node:os"; import * as path from "node:path"; import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; @@ -8,6 +10,7 @@ import { $which, ptree, truncate } from "@oh-my-pi/pi-utils"; import { parseHTML } from "linkedom"; import { LRUCache } from "lru-cache/raw"; import type { Settings } from "../config/settings"; +import { readEditableNotebookText } from "../edit/notebook"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { type Theme, theme } from "../modes/theme/theme"; import type { ToolSession } from "../sdk"; @@ -22,10 +25,12 @@ import { specialHandlers } from "../web/scrapers"; import type { RenderResult } from "../web/scrapers/types"; import { finalizeOutput, loadPage, looksLikeHtml, MAX_OUTPUT_CHARS } from "../web/scrapers/types"; import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils"; +import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "./archive-reader"; import { applyListLimit } from "./list-limit"; import { formatStyledArtifactReference, type OutputMeta } from "./output-meta"; import { type LineRange, parseLineRanges } from "./path-utils"; -import { formatExpandHint, getDomain, replaceTabs } from "./render-utils"; +import { formatBytes, formatExpandHint, getDomain, replaceTabs } from "./render-utils"; +import { listTables, looksLikeSqlite, renderTableList } from "./sqlite-reader"; import { ToolAbortError, ToolError } from "./tool-errors"; import { toolResult } from "./tool-result"; import { clampTimeout } from "./tool-timeouts"; @@ -46,8 +51,6 @@ const CONVERTIBLE_MIMES = new Set([ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "application/rtf", "application/epub+zip", - "application/x-ipynb+json", - "application/zip", "image/png", "image/jpeg", "image/gif", @@ -67,7 +70,6 @@ const CONVERTIBLE_EXTENSIONS = new Set([ ".xlsx", ".rtf", ".epub", - ".ipynb", ".png", ".jpg", ".jpeg", @@ -78,6 +80,27 @@ const CONVERTIBLE_EXTENSIONS = new Set([ ".ogg", ]); +const NOTEBOOK_MIMES = new Set(["application/x-ipynb+json"]); +const NOTEBOOK_EXTENSIONS = new Set([".ipynb"]); + +const SQLITE_MIMES = new Set([ + "application/vnd.sqlite3", + "application/x-sqlite3", + "application/sqlite3", + "application/sqlite", +]); +const SQLITE_EXTENSIONS = new Set([".sqlite", ".sqlite3", ".db", ".db3"]); + +const ARCHIVE_MIMES = new Set([ + "application/zip", + "application/x-zip-compressed", + "application/x-tar", + "application/tar", + "application/gzip", + "application/x-gzip", +]); +const ARCHIVE_EXTENSIONS = new Set([".zip", ".tar", ".tar.gz", ".tgz", ".gz"]); + const IMAGE_MIME_BY_EXTENSION = new Map([ [".png", "image/png"], [".jpg", "image/jpeg"], @@ -261,6 +284,12 @@ function normalizeMime(contentType: string): string { return contentType.split(";")[0].trim().toLowerCase(); } +function getFilenameExtensionHint(filename: string): string { + const lower = filename.toLowerCase(); + if (lower.endsWith(".tar.gz")) return ".tar.gz"; + return path.extname(filename).toLowerCase(); +} + /** * Get extension from URL or Content-Disposition */ @@ -269,7 +298,7 @@ function getExtensionHint(url: string, contentDisposition?: string): string { if (contentDisposition) { const match = contentDisposition.match(/filename[*]?=["']?([^"';\n]+)/i); if (match) { - const ext = path.extname(match[1]).toLowerCase(); + const ext = getFilenameExtensionHint(match[1]); if (ext) return ext; } } @@ -277,7 +306,7 @@ function getExtensionHint(url: string, contentDisposition?: string): string { // Fall back to URL path try { const pathname = new URL(url).pathname; - const ext = path.extname(pathname).toLowerCase(); + const ext = getFilenameExtensionHint(pathname); if (ext) return ext; } catch {} @@ -738,6 +767,254 @@ type FetchRenderResult = RenderResult & { image?: FetchImagePayload; }; +const BINARY_SAMPLE_CHARS = 4096; +const URL_ARCHIVE_LIST_LIMIT = 500; +const URL_SQLITE_LIST_LIMIT = 500; + +function sampleLooksBinary(text: string): boolean { + const limit = Math.min(text.length, BINARY_SAMPLE_CHARS); + if (limit === 0) return false; + + let replacementCount = 0; + for (let index = 0; index < limit; index++) { + const code = text.charCodeAt(index); + if (code === 0) return true; + if (code === 0xfffd) replacementCount++; + } + + return replacementCount >= 3 && replacementCount / limit > 0.01; +} + +function isNotebookHint(mime: string, extensionHint: string): boolean { + return NOTEBOOK_MIMES.has(mime) || NOTEBOOK_EXTENSIONS.has(extensionHint); +} + +function isSqliteHint(mime: string, extensionHint: string): boolean { + return SQLITE_MIMES.has(mime) || SQLITE_EXTENSIONS.has(extensionHint); +} + +function isArchiveHint(mime: string, extensionHint: string): boolean { + return ARCHIVE_MIMES.has(mime) || ARCHIVE_EXTENSIONS.has(extensionHint); +} + +function getArchiveFormatHint(mime: string, extensionHint: string): ArchiveFormat | undefined { + if (extensionHint === ".zip" || mime === "application/zip" || mime === "application/x-zip-compressed") { + return "zip"; + } + if (extensionHint === ".tar" || mime === "application/x-tar" || mime === "application/tar") { + return "tar"; + } + if ( + extensionHint === ".tar.gz" || + extensionHint === ".tgz" || + extensionHint === ".gz" || + mime === "application/gzip" || + mime === "application/x-gzip" + ) { + return "tar.gz"; + } + return undefined; +} + +function formatErrorMessage(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +function binaryContentType(mime: string): string { + return mime || "application/octet-stream"; +} + +function buildBinaryNotice(finalUrl: string, mime: string, byteLength?: number): string { + const size = byteLength === undefined ? "unknown size" : formatBytes(byteLength); + return `[Binary content: ${binaryContentType(mime)}, ${size}] ${finalUrl}`; +} + +function buildBinaryPayloadResult( + url: string, + finalUrl: string, + mime: string, + method: string, + content: string, + fetchedAt: string, + notes: string[], +): FetchRenderResult { + const output = finalizeOutput(content); + return { + url, + finalUrl, + contentType: binaryContentType(mime), + method, + content: output.content, + fetchedAt, + truncated: output.truncated, + notes, + }; +} + +async function withTempBinaryFile( + prefix: string, + extension: string, + bytes: Uint8Array, + readTempFile: (tempPath: string) => Promise, +): Promise { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), prefix)); + const tempPath = path.join(tempDir, `payload${extension}`); + try { + await Bun.write(tempPath, bytes); + return await readTempFile(tempPath); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } +} + +async function renderNotebookPayload(bytes: Uint8Array, displayUrl: string): Promise { + return withTempBinaryFile("omp-url-notebook-", ".ipynb", bytes, tempPath => + readEditableNotebookText(tempPath, displayUrl), + ); +} + +async function renderSqlitePayload(bytes: Uint8Array): Promise { + return withTempBinaryFile("omp-url-sqlite-", ".sqlite", bytes, async tempPath => { + let db: Database | null = null; + try { + db = new Database(tempPath, { readonly: true, strict: true }); + db.run("PRAGMA busy_timeout = 3000"); + const listLimit = applyListLimit(listTables(db), { limit: URL_SQLITE_LIST_LIMIT }); + return renderTableList(listLimit.items); + } finally { + db?.close(); + } + }); +} + +async function tryRenderBinaryPayload( + url: string, + finalUrl: string, + mime: string, + extHint: string, + rawContent: string, + timeout: number, + signal: AbortSignal | undefined, + fetchedAt: string, + notes: readonly string[], +): Promise { + const hasNotebookHint = isNotebookHint(mime, extHint); + const hasSqliteHint = isSqliteHint(mime, extHint); + const hasArchiveHint = isArchiveHint(mime, extHint); + const rawLooksBinary = sampleLooksBinary(rawContent); + if (!hasNotebookHint && !hasSqliteHint && !hasArchiveHint && !rawLooksBinary) { + return null; + } + + const resultNotes = [...notes]; + const binary = await fetchBinary(finalUrl, timeout, signal); + if (!binary.ok) { + resultNotes.push(binary.error ? `Binary fetch failed: ${binary.error}` : "Binary fetch failed"); + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "binary", + buildBinaryNotice(finalUrl, mime), + fetchedAt, + resultNotes, + ); + } + + const binaryExtHint = getExtensionHint(finalUrl, binary.contentDisposition) || extHint; + if (isNotebookHint(mime, binaryExtHint)) { + try { + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "notebook", + await renderNotebookPayload(binary.buffer, finalUrl), + fetchedAt, + resultNotes, + ); + } catch (error) { + resultNotes.push(`Notebook rendering failed: ${formatErrorMessage(error)}`); + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "binary", + buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength), + fetchedAt, + resultNotes, + ); + } + } + + if (isSqliteHint(mime, binaryExtHint) || looksLikeSqlite(binary.buffer)) { + try { + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "sqlite", + await renderSqlitePayload(binary.buffer), + fetchedAt, + resultNotes, + ); + } catch (error) { + resultNotes.push(`SQLite rendering failed: ${formatErrorMessage(error)}`); + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "binary", + buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength), + fetchedAt, + resultNotes, + ); + } + } + + const hintedArchiveFormat = getArchiveFormatHint(mime, binaryExtHint); + const shouldArchiveSniff = hintedArchiveFormat !== undefined || !isConvertible(mime, binaryExtHint); + const archiveFormat = hintedArchiveFormat ?? (shouldArchiveSniff ? sniffArchiveFormat(binary.buffer) : undefined); + if (archiveFormat) { + try { + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "archive", + await listArchiveRoot(binary.buffer, archiveFormat, { limit: URL_ARCHIVE_LIST_LIMIT }), + fetchedAt, + resultNotes, + ); + } catch (error) { + resultNotes.push(`Archive rendering failed: ${formatErrorMessage(error)}`); + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "binary", + buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength), + fetchedAt, + resultNotes, + ); + } + } + + if (rawLooksBinary) { + return buildBinaryPayloadResult( + url, + finalUrl, + mime, + "binary", + buildBinaryNotice(finalUrl, mime, binary.buffer.byteLength), + fetchedAt, + resultNotes, + ); + } + + return null; +} + // ============================================================================= // Unified Special Handler Dispatch // ============================================================================= @@ -984,6 +1261,19 @@ async function renderUrl( } } + const binaryPayloadResult = await tryRenderBinaryPayload( + url, + finalUrl, + mime, + extHint, + rawContent, + timeout, + signal, + fetchedAt, + notes, + ); + if (binaryPayloadResult) return binaryPayloadResult; + // Step 4: Handle non-HTML text content const isHtml = mime.includes("html") || mime.includes("xhtml"); const isJson = mime.includes("json"); @@ -992,7 +1282,7 @@ async function renderUrl( const isFeed = mime.includes("rss") || mime.includes("atom") || mime.includes("feed"); // Raw mode skips every text-shaping branch below (JSON pretty-print, feed-to-markdown, - // HTML extraction) and returns the response body verbatim. The image/markit branches + // HTML extraction) and returns the response body verbatim. Binary-oriented branches // above already ran because raw isn't useful for binary payloads. if (raw) { const output = finalizeOutput(rawContent); diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 48df2758b..020ac82b4 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -34,7 +34,7 @@ import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { ImageInputTooLargeError, loadImageInput, MAX_IMAGE_INPUT_BYTES } from "../utils/image-loading"; import { convertFileWithMarkit } from "../utils/markit"; import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree"; -import { type ArchiveReader, openArchive, parseArchivePathCandidates } from "./archive-reader"; +import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader"; import { type ConflictEntry, type ConflictScope, @@ -1154,17 +1154,10 @@ export class ReadTool implements AgentTool { const limitedEntries = listLimit.items; const limitMeta = listLimit.meta; - const results: string[] = []; - for (const entry of limitedEntries) { + for (let index = 0; index < limitedEntries.length; index++) { throwIfAborted(signal); - if (entry.isDirectory) { - results.push(`${entry.name}/`); - continue; - } - - const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : ""; - results.push(`${entry.name}${sizeSuffix}`); } + const results = formatArchiveEntryLines(limitedEntries); const output = results.length > 0 ? results.join("\n") : "(empty archive directory)"; const text = prependSuffixResolutionNotice(output, details.suffixResolution); diff --git a/packages/coding-agent/src/tools/sqlite-reader.ts b/packages/coding-agent/src/tools/sqlite-reader.ts index 37948a48d..710c7fa66 100644 --- a/packages/coding-agent/src/tools/sqlite-reader.ts +++ b/packages/coding-agent/src/tools/sqlite-reader.ts @@ -5,6 +5,14 @@ import { ToolError } from "./tool-errors"; const SQLITE_MAGIC = new Uint8Array([ 0x53, 0x51, 0x4c, 0x69, 0x74, 0x65, 0x20, 0x66, 0x6f, 0x72, 0x6d, 0x61, 0x74, 0x20, 0x33, 0x00, ]); + +export function looksLikeSqlite(bytes: Uint8Array): boolean { + if (bytes.byteLength < SQLITE_MAGIC.byteLength) return false; + for (const [index, byte] of SQLITE_MAGIC.entries()) { + if (bytes[index] !== byte) return false; + } + return true; +} const SQLITE_PATH_PATTERN = /\.(?:sqlite3?|db3?)(?=(?::|\?|$))/gi; const DEFAULT_QUERY_LIMIT = 20; const DEFAULT_SCHEMA_SAMPLE_LIMIT = 5; @@ -443,18 +451,7 @@ export function parseSqlitePathCandidates(filePath: string): SqlitePathCandidate export async function isSqliteFile(absolutePath: string): Promise { try { - const bytes = await Bun.file(absolutePath).slice(0, SQLITE_MAGIC.byteLength).bytes(); - if (bytes.length !== SQLITE_MAGIC.byteLength) { - return false; - } - - for (const [index, byte] of SQLITE_MAGIC.entries()) { - if (bytes[index] !== byte) { - return false; - } - } - - return true; + return looksLikeSqlite(await Bun.file(absolutePath).slice(0, SQLITE_MAGIC.byteLength).bytes()); } catch { return false; } diff --git a/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts b/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts new file mode 100644 index 000000000..cab602c17 --- /dev/null +++ b/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts @@ -0,0 +1,217 @@ +import { Database } from "bun:sqlite"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; +import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types"; +import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { zipSync } from "fflate"; + +function makeSession(testDir: string): ToolSession { + const sessionFile = path.join(testDir, "session.jsonl"); + const artifactsDir = sessionFile.slice(0, -6); + let nextArtifactId = 0; + return { + cwd: testDir, + hasUI: false, + getSessionFile: () => sessionFile, + getArtifactsDir: () => artifactsDir, + getSessionSpawns: () => null, + allocateOutputArtifact: async toolType => { + const id = String(nextArtifactId++); + return { id, path: path.join(artifactsDir, `${id}.${toolType}.log`) }; + }, + settings: Settings.isolated({ "fetch.enabled": true }), + }; +} + +function stubUrlBytes(bytes: Uint8Array, contentType: string, contentDisposition?: string) { + const decoded = Buffer.from(bytes).toString("utf-8"); + vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({ + ok: true, + status: 200, + finalUrl: requestedUrl, + contentType, + content: decoded, + })); + return vi.spyOn(scraperUtils, "fetchBinary").mockImplementation(async () => ({ + ok: true, + buffer: bytes, + contentDisposition, + })); +} + +function stubUrlText(body: string, contentType: string) { + vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({ + ok: true, + status: 200, + finalUrl: requestedUrl, + contentType, + content: body, + })); + return vi.spyOn(scraperUtils, "fetchBinary").mockImplementation(async () => { + throw new Error("unexpected binary fetch"); + }); +} + +function textOutput(result: { content: Array }): string { + return result.content + .filter((content): content is TextContent => content.type === "text") + .map(content => content.text) + .join("\n"); +} + +async function createSqliteFixtureBytes(testDir: string): Promise { + const dbPath = path.join(testDir, `url-fixture-${Snowflake.next()}.sqlite`); + const db = new Database(dbPath); + try { + db.exec(` + CREATE TABLE notes (id INTEGER PRIMARY KEY, title TEXT NOT NULL); + INSERT INTO notes (title) VALUES ('alpha'), ('beta'); + `); + } finally { + db.close(); + } + return Bun.file(dbPath).bytes(); +} + +function createNotebookFixtureBytes(): Uint8Array { + return Buffer.from( + JSON.stringify({ + cells: [ + { cell_type: "markdown", metadata: {}, source: ["# Remote notebook\n", "Rendered as editable cells"] }, + { cell_type: "code", execution_count: null, metadata: {}, outputs: [], source: ["answer = 42\n"] }, + ], + metadata: {}, + nbformat: 4, + nbformat_minor: 5, + }), + ); +} + +function uniqueUrl(name: string, extension: string): string { + return `https://example.com/${name}-${Snowflake.next()}${extension}`; +} + +describe("read URL binary dispatch", () => { + let testDir: string; + + beforeEach(() => { + testDir = path.join(os.tmpdir(), `fetch-binary-dispatch-${Snowflake.next()}`); + fs.mkdirSync(testDir, { recursive: true }); + }); + + afterEach(() => { + vi.restoreAllMocks(); + fs.rmSync(testDir, { recursive: true, force: true }); + }); + + it("lists a remote zip instead of dumping decoded bytes", async () => { + const zipBytes = zipSync({ + "root.txt": Buffer.from("root file\n"), + "nested/data.txt": Buffer.from("nested file\n"), + }); + const url = uniqueUrl("archive", ".zip"); + stubUrlBytes(zipBytes, "application/octet-stream"); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-zip", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("archive"); + expect(text).toContain("Method: archive"); + expect(text).toContain("root.txt"); + expect(text).toContain("nested/"); + expect(text).not.toContain("PK\u0003\u0004"); + expect(text).not.toContain("�"); + }); + + it("returns a metadata notice when a hinted binary refetch fails", async () => { + const url = uniqueUrl("oversized", ".zip"); + vi.spyOn(scrapers, "loadPage").mockImplementation(async requestedUrl => ({ + ok: true, + status: 200, + finalUrl: requestedUrl, + contentType: "application/octet-stream", + content: "PK\u0003\u0004\u0000\u0001", + })); + vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({ + ok: false, + error: "content-length 52428801 exceeds 52428800", + }); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-zip-too-large", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("binary"); + expect(result.details?.notes).toContain("Binary fetch failed: content-length 52428801 exceeds 52428800"); + expect(text).toContain("[Binary content: application/octet-stream"); + expect(text).not.toContain("PK\u0003\u0004"); + }); + + it("renders a remote sqlite database through the sqlite reader", async () => { + const sqliteBytes = await createSqliteFixtureBytes(testDir); + const url = uniqueUrl("data", ".db"); + stubUrlBytes(sqliteBytes, "application/octet-stream"); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-sqlite", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("sqlite"); + expect(text).toContain("Method: sqlite"); + expect(text).toContain("notes (2 rows)"); + }); + + it("renders a remote notebook as editable cells", async () => { + const notebookBytes = createNotebookFixtureBytes(); + const url = uniqueUrl("notebook", ".ipynb"); + stubUrlBytes(notebookBytes, "application/octet-stream"); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-notebook", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("notebook"); + expect(text).toContain("Method: notebook"); + expect(text).toContain("# %% [markdown] cell:0"); + expect(text).toContain("# %% [code] cell:1"); + expect(text).toContain("answer = 42"); + }); + + it("returns a metadata notice for unrenderable binary bytes", async () => { + const binaryBytes = new Uint8Array([0, 1, 2, 3, 255, 254, 253, 0, 7, 8]); + const url = uniqueUrl("payload", ".bin"); + stubUrlBytes(binaryBytes, "application/octet-stream"); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-binary", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("binary"); + expect(text).toContain("Method: binary"); + expect(text).toContain("[Binary content: application/octet-stream"); + expect(text).not.toContain("\u0000"); + expect(text).not.toContain("�"); + }); + + it("leaves valid UTF-8 octet-stream payloads on the text path", async () => { + const url = uniqueUrl("plain", ".txt"); + const fetchBinarySpy = stubUrlText("plain UTF-8 text\nsecond line", "application/octet-stream"); + + const tool = new ReadTool(makeSession(testDir)); + const result = await tool.execute("read-url-text-octet", { path: url }); + const text = textOutput(result); + + expect(result.details?.method).toBe("raw"); + expect(text).toContain("plain UTF-8 text"); + expect(text).toContain("second line"); + expect(fetchBinarySpy).not.toHaveBeenCalled(); + }); +});