diff --git a/docs/tools/read.md b/docs/tools/read.md index f73bd34bd..4df731f42 100644 --- a/docs/tools/read.md +++ b/docs/tools/read.md @@ -7,7 +7,7 @@ - Model-facing prompt: `packages/coding-agent/src/prompts/tools/read.md` - Key collaborators: - `packages/coding-agent/src/tools/path-utils.ts` — split `path` from trailing selectors; normalize local paths. - - `packages/coding-agent/src/tools/archive-reader.ts` — detect `archive.ext:inner/path`, index archives, list/read entries. + - `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: detect `archive.ext:inner/path`, index archives, list/read entries. - `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite targets, parse selectors, render tables. - `packages/coding-agent/src/tools/fetch.ts` — URL parsing, fetch/render pipeline, URL cache/artifacts. - `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`. diff --git a/docs/tools/write.md b/docs/tools/write.md index b37859e39..23c366d11 100644 --- a/docs/tools/write.md +++ b/docs/tools/write.md @@ -6,7 +6,7 @@ - Entry: `packages/coding-agent/src/tools/write.ts` - Model-facing prompt: `packages/coding-agent/src/prompts/tools/write.md` - Key collaborators: - - `packages/coding-agent/src/tools/archive-reader.ts` — parse `archive.ext:entry` selectors. + - `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: parse `archive.ext:entry` selectors and rewrite the archive whole. - `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite paths and perform row insert/update/delete. - `packages/coding-agent/src/tools/conflict-detect.ts` — parse `conflict://` URIs and splice recorded merge-conflict regions. - `packages/coding-agent/src/lsp/index.ts` — format-on-write and diagnostics writethrough. @@ -56,7 +56,7 @@ Single-shot result. 1. `WriteTool.execute()` in `packages/coding-agent/src/tools/write.ts` strips pasted `[PATH#HASH]` headers and `LINE:` hashline prefixes from `content` when the session is in hashline display mode. 2. If `path` is an internal URL whose handler exposes `write`, the tool delegates directly to `handler.write(...)` and returns. 3. `conflict://...` paths are handled next by the merge-conflict resolver. Scope reads such as `conflict:///ours` are rejected as read-only; writable conflict URIs must omit the scope. -4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/tools/archive-reader.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet. +4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/utils/zip.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet. 5. Archive writes call `enforcePlanModeWrite(..., { op: exists ? "update" : "create" })`, then `#writeArchiveEntry()`. - The parent directory of the archive file is created with `fs.mkdir(..., { recursive: true })`. - `.zip` archives are read with `fflate.unzipSync()`, the target entry is replaced in an in-memory map, and the archive is rewritten with `fflate.zipSync()` + `Bun.write()`. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 66fd5ffdf..9833f84e6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,11 @@ # Changelog ## [Unreleased] - ### Changed +- Refactored internal archive handling into a unified `src/utils/zip.ts` module +- Centralized all `fflate` (ZIP) and `Bun.Archive` (tar/tar.gz) operations into `zip.ts` +- Optimized archive reading by using lazy, ranged central-directory access for ZIP files - Updated internal image processing to no longer include metadata text for fetched images - Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob - Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved. diff --git a/packages/coding-agent/src/debug/report-bundle.ts b/packages/coding-agent/src/debug/report-bundle.ts index 0e7914c47..f633ca64b 100644 --- a/packages/coding-agent/src/debug/report-bundle.ts +++ b/packages/coding-agent/src/debug/report-bundle.ts @@ -7,6 +7,7 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; import type { WorkProfile } from "@oh-my-pi/pi-natives"; import { APP_NAME, getLogPath, getLogsDir, getReportsDir, isEnoent } from "@oh-my-pi/pi-utils"; +import { writeArchive } from "../utils/zip"; import type { CpuProfile, HeapSnapshot } from "./profiler"; import { collectSystemInfo, sanitizeEnv } from "./system-info"; @@ -165,7 +166,7 @@ export async function createReportBundle(options: ReportBundleOptions): Promise< } // Write archive - await Bun.Archive.write(outputPath, data, { compress: "gzip" }); + await writeArchive(outputPath, "tar.gz", Object.entries(data)); return { path: outputPath, files }; } diff --git a/packages/coding-agent/src/tools/archive-reader.ts b/packages/coding-agent/src/tools/archive-reader.ts deleted file mode 100644 index 5b44c855a..000000000 --- a/packages/coding-agent/src/tools/archive-reader.ts +++ /dev/null @@ -1,721 +0,0 @@ -import * as fs from "node:fs/promises"; -import * as os from "node:os"; -import * as path from "node:path"; -import { bytesToText, inflateRaw } from "../utils/zip"; - -import { formatBytes } from "./render-utils"; -import { ToolError } from "./tool-errors"; - -/** - * Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into - * memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is - * exempt: it is read via ranged central-directory access. - */ -const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024; -/** - * Cap on a single archive member's declared (uncompressed) size. The declared - * size is attacker-controlled metadata — a crafted ZIP entry can claim - * multi-GB sizes that would be allocated up front before any data inflates. - */ -const MAX_ARCHIVE_MEMBER_BYTES = 64 * 1024 * 1024; - -export type ArchiveFormat = "zip" | "tar" | "tar.gz"; - -export interface ArchivePathCandidate { - archivePath: string; - subPath: string; -} - -export interface ArchiveNode { - path: string; - isDirectory: boolean; - size: number; - mtimeMs?: number; -} - -export interface ArchiveDirectoryEntry extends ArchiveNode { - name: string; -} - -export interface ExtractedArchiveFile extends ArchiveNode { - bytes: Uint8Array; -} - -interface TarStorage { - type: "tar"; - file: File; -} - -interface ZipStorage { - type: "zip"; - archivePath: string; - compressedSize: number; - compression: number; - flags: number; - localHeaderOffset: number; -} - -type EntryStorage = TarStorage | ZipStorage; - -interface ArchiveIndexEntry extends ArchiveNode { - storage?: EntryStorage; -} - -function normalizeArchiveLookupPath(rawPath?: string): string | undefined { - if (!rawPath) return ""; - - const parts = rawPath.replace(/\\/g, "/").split("/"); - const normalizedParts: string[] = []; - for (const part of parts) { - if (!part || part === ".") continue; - if (part === "..") return undefined; - normalizedParts.push(part); - } - - return normalizedParts.join("/"); -} - -function normalizeArchiveEntryPath(rawPath: string): string | undefined { - const parts = rawPath.replace(/\\/g, "/").split("/"); - const normalizedParts: string[] = []; - for (const part of parts) { - if (!part || part === ".") continue; - if (part === "..") return undefined; - normalizedParts.push(part); - } - - if (normalizedParts.length === 0) return undefined; - return normalizedParts.join("/"); -} - -function isArchiveDirectoryName(rawPath: string): boolean { - return rawPath.endsWith("/") || rawPath.endsWith("\\"); -} - -function upsertArchiveEntry(map: Map, entry: ArchiveIndexEntry): void { - const existing = map.get(entry.path); - if (!existing) { - map.set(entry.path, entry); - return; - } - - if (existing.isDirectory && !entry.isDirectory) { - map.set(entry.path, entry); - return; - } - - if (!existing.isDirectory && entry.isDirectory) { - return; - } - - map.set(entry.path, { - ...existing, - size: existing.size || entry.size, - mtimeMs: existing.mtimeMs ?? entry.mtimeMs, - storage: existing.storage ?? entry.storage, - }); -} - -function ensureParentDirectories(map: Map): void { - for (const entry of [...map.values()]) { - const parts = entry.path.split("/"); - const stop = parts.length - 1; - for (let index = 1; index <= stop; index++) { - const dirPath = parts.slice(0, index).join("/"); - if (!dirPath || map.has(dirPath)) continue; - map.set(dirPath, { - path: dirPath, - isDirectory: true, - size: 0, - }); - } - } -} - -function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined { - const normalized = filePath.toLowerCase(); - if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz"; - if (normalized.endsWith(".tar")) return "tar"; - if (normalized.endsWith(".zip")) return "zip"; - return undefined; -} - -export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] { - return entries.map(entry => { - if (entry.isDirectory) return `${entry.name}/`; - - const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : ""; - return `${entry.name}${sizeSuffix}`; - }); -} - -const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50; -const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50; -const ZIP64_EOCD_SIGNATURE = 0x06064b50; -const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50; -const ZIP_EOCD_SIGNATURE = 0x06054b50; -const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50; -const ZIP_EOCD_MIN_LENGTH = 22; -const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff; -const ZIP64_EOCD_LOCATOR_LENGTH = 20; -const ZIP_STORED_COMPRESSION = 0; -const ZIP_DEFLATE_COMPRESSION = 8; -const ZIP_UTF8_FLAG = 0x0800; -const ZIP_ENCRYPTED_FLAG = 0x0001; -const ZIP_UINT16_MAX = 0xffff; -const ZIP_UINT32_MAX = 0xffffffff; -const ZIP_UINT32_RANGE = 0x100000000; - -interface ZipCentralDirectoryInfo { - entries: number; - offset: number; - size: number; -} - -interface Zip64EntryValues { - compressedSize: number; - uncompressedSize: number; - localHeaderOffset: number; - diskStart: number; -} - -interface Zip64EntryPlaceholders { - compressedSize: boolean; - uncompressedSize: boolean; - localHeaderOffset: boolean; - diskStart: boolean; -} - -function readUInt16LE(bytes: Uint8Array, offset: number): number { - return bytes[offset]! | (bytes[offset + 1]! << 8); -} - -function readUInt32LE(bytes: Uint8Array, offset: number): number { - return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0; -} - -function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean { - if (bytes.byteLength < offset + value.length) return false; - for (let index = 0; index < value.length; index++) { - if (bytes[offset + index] !== value.charCodeAt(index)) return false; - } - return true; -} - -export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined { - if (bytes.byteLength >= 4) { - const signature = readUInt32LE(bytes, 0); - if ( - signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE || - signature === ZIP_EOCD_SIGNATURE || - signature === ZIP_DATA_DESCRIPTOR_SIGNATURE - ) { - return "zip"; - } - } - - if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) { - return "tar.gz"; - } - - if (bytesMatchAscii(bytes, 257, "ustar")) { - return "tar"; - } - - return undefined; -} - -function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number { - const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE; - if (!Number.isSafeInteger(value)) { - throw new ToolError("ZIP archive uses offsets or sizes too large to read safely"); - } - return value; -} - -async function readZipRange(filePath: string, start: number, end: number): Promise { - if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) { - throw new ToolError("Invalid ZIP archive range"); - } - - const bytes = await Bun.file(filePath).slice(start, end).bytes(); - if (bytes.byteLength !== end - start) { - throw new ToolError("Invalid ZIP archive: truncated data"); - } - return bytes; -} - -function findEndOfCentralDirectory(tail: Uint8Array): number { - for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) { - if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue; - const commentLength = readUInt16LE(tail, offset + 20); - if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset; - } - - throw new ToolError("Invalid ZIP archive: missing end of central directory"); -} - -async function readZip64CentralDirectoryInfo( - filePath: string, - tail: Uint8Array, - tailStart: number, - eocdOffset: number, -): Promise { - const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH; - if (locatorOffset < 0) return undefined; - - const locator = - locatorOffset >= tailStart - ? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH) - : await readZipRange(filePath, locatorOffset, eocdOffset); - if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined; - - const zip64EocdDisk = readUInt32LE(locator, 4); - const zip64EocdOffset = readUInt64LEAsNumber(locator, 8); - const totalDisks = readUInt32LE(locator, 16); - if (zip64EocdDisk !== 0 || totalDisks > 1) { - throw new ToolError("Multi-disk ZIP archives are not supported"); - } - - const record = await readZipRange(filePath, zip64EocdOffset, zip64EocdOffset + 56); - if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) { - throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory"); - } - if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) { - throw new ToolError("Multi-disk ZIP archives are not supported"); - } - - return { - entries: readUInt64LEAsNumber(record, 32), - size: readUInt64LEAsNumber(record, 40), - offset: readUInt64LEAsNumber(record, 48), - }; -} - -async function readZipCentralDirectoryInfo(filePath: string, fileSize: number): Promise { - if (fileSize < ZIP_EOCD_MIN_LENGTH) { - throw new ToolError("Invalid ZIP archive: missing end of central directory"); - } - - const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH); - const tailStart = fileSize - tailLength; - const tail = await readZipRange(filePath, tailStart, fileSize); - const eocdIndex = findEndOfCentralDirectory(tail); - const eocdOffset = tailStart + eocdIndex; - - if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) { - throw new ToolError("Multi-disk ZIP archives are not supported"); - } - - let entries = readUInt16LE(tail, eocdIndex + 10); - let size = readUInt32LE(tail, eocdIndex + 12); - let offset = readUInt32LE(tail, eocdIndex + 16); - const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX; - const zip64Info = await readZip64CentralDirectoryInfo(filePath, tail, tailStart, eocdOffset); - if (zip64Info) { - ({ entries, size, offset } = zip64Info); - } else if (needsZip64) { - throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata"); - } - - if (offset + size > fileSize) { - throw new ToolError("Invalid ZIP archive: central directory exceeds file size"); - } - - return { entries, offset, size }; -} - -function readZip64EntryValues( - extra: Uint8Array, - placeholders: Zip64EntryPlaceholders, - current: Zip64EntryValues, -): Zip64EntryValues { - if ( - !placeholders.compressedSize && - !placeholders.uncompressedSize && - !placeholders.localHeaderOffset && - !placeholders.diskStart - ) { - return current; - } - - let offset = 0; - while (offset + 4 <= extra.byteLength) { - const headerId = readUInt16LE(extra, offset); - const dataSize = readUInt16LE(extra, offset + 2); - const dataStart = offset + 4; - const dataEnd = dataStart + dataSize; - if (dataEnd > extra.byteLength) { - throw new ToolError("Invalid ZIP archive: malformed extra field"); - } - - if (headerId === 0x0001) { - let cursor = dataStart; - let uncompressedSize = current.uncompressedSize; - let compressedSize = current.compressedSize; - let localHeaderOffset = current.localHeaderOffset; - let diskStart = current.diskStart; - - if (placeholders.uncompressedSize) { - if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); - uncompressedSize = readUInt64LEAsNumber(extra, cursor); - cursor += 8; - } - if (placeholders.compressedSize) { - if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); - compressedSize = readUInt64LEAsNumber(extra, cursor); - cursor += 8; - } - if (placeholders.localHeaderOffset) { - if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); - localHeaderOffset = readUInt64LEAsNumber(extra, cursor); - cursor += 8; - } - if (placeholders.diskStart) { - if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); - diskStart = readUInt32LE(extra, cursor); - } - - return { compressedSize, uncompressedSize, localHeaderOffset, diskStart }; - } - - offset = dataEnd; - } - - throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field"); -} - -function parseZipCentralDirectory( - filePath: string, - centralDirectory: Uint8Array, - expectedEntries: number, -): ArchiveIndexEntry[] { - const entries: ArchiveIndexEntry[] = []; - let offset = 0; - - for (let index = 0; index < expectedEntries; index++) { - if (offset + 46 > centralDirectory.byteLength) { - throw new ToolError("Invalid ZIP archive: truncated central directory"); - } - if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) { - throw new ToolError("Invalid ZIP archive: malformed central directory"); - } - - const flags = readUInt16LE(centralDirectory, offset + 8); - const compression = readUInt16LE(centralDirectory, offset + 10); - const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20); - const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24); - const fileNameLength = readUInt16LE(centralDirectory, offset + 28); - const extraLength = readUInt16LE(centralDirectory, offset + 30); - const commentLength = readUInt16LE(centralDirectory, offset + 32); - const diskStartRaw = readUInt16LE(centralDirectory, offset + 34); - const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42); - const nameStart = offset + 46; - const extraStart = nameStart + fileNameLength; - const entryEnd = extraStart + extraLength + commentLength; - if (entryEnd > centralDirectory.byteLength) { - throw new ToolError("Invalid ZIP archive: truncated central directory entry"); - } - - const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0); - const normalizedPath = normalizeArchiveEntryPath(rawPath); - if (normalizedPath) { - const values = readZip64EntryValues( - centralDirectory.subarray(extraStart, extraStart + extraLength), - { - compressedSize: compressedSizeRaw === ZIP_UINT32_MAX, - uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX, - localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX, - diskStart: diskStartRaw === ZIP_UINT16_MAX, - }, - { - compressedSize: compressedSizeRaw, - uncompressedSize: uncompressedSizeRaw, - localHeaderOffset: localHeaderOffsetRaw, - diskStart: diskStartRaw, - }, - ); - if (values.diskStart !== 0) { - throw new ToolError("Multi-disk ZIP archives are not supported"); - } - - const isDirectory = isArchiveDirectoryName(rawPath); - entries.push({ - path: normalizedPath, - isDirectory, - size: isDirectory ? 0 : values.uncompressedSize, - storage: isDirectory - ? undefined - : { - type: "zip", - archivePath: filePath, - compressedSize: values.compressedSize, - compression, - flags, - localHeaderOffset: values.localHeaderOffset, - }, - }); - } - - offset = entryEnd; - } - - return entries; -} - -async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise { - if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) { - throw new ToolError("Encrypted ZIP entries are not supported"); - } - - const localHeader = await readZipRange( - storage.archivePath, - storage.localHeaderOffset, - storage.localHeaderOffset + 30, - ); - if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) { - throw new ToolError("Invalid ZIP archive: malformed local file header"); - } - - const fileNameLength = readUInt16LE(localHeader, 26); - const extraLength = readUInt16LE(localHeader, 28); - const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength; - const compressedBytes = await readZipRange(storage.archivePath, dataStart, dataStart + storage.compressedSize); - - if (storage.compression === ZIP_STORED_COMPRESSION) { - return compressedBytes; - } - if (storage.compression !== ZIP_DEFLATE_COMPRESSION) { - throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`); - } - - try { - return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize)); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } -} - -async function readTarEntries(bytes: Uint8Array): Promise { - let archive: Bun.Archive; - try { - archive = new Bun.Archive(bytes); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } - - let files: Map; - try { - files = await archive.files(); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } - - const entries: ArchiveIndexEntry[] = []; - for (const [rawPath, file] of files) { - const normalizedPath = normalizeArchiveEntryPath(rawPath); - if (!normalizedPath) continue; - const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined; - entries.push({ - path: normalizedPath, - isDirectory: false, - size: file.size, - mtimeMs, - storage: { type: "tar", file }, - }); - } - - return entries; -} - -async function readZipEntries(filePath: string): Promise { - const fileSize = Bun.file(filePath).size; - if (!Number.isSafeInteger(fileSize)) { - throw new ToolError("ZIP archive is too large to read safely"); - } - - const directoryInfo = await readZipCentralDirectoryInfo(filePath, fileSize); - const centralDirectory = await readZipRange( - filePath, - directoryInfo.offset, - directoryInfo.offset + directoryInfo.size, - ); - return parseZipCentralDirectory(filePath, centralDirectory, directoryInfo.entries); -} - -export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] { - const normalized = filePath.replace(/\\/g, "/"); - const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi; - const seen = new Set(); - const candidates: ArchivePathCandidate[] = []; - - let match: RegExpExecArray | null; - while (true) { - match = pattern.exec(normalized); - if (match === null) { - break; - } - const end = match.index + match[0].length; - const archivePath = filePath.slice(0, end); - const subPath = normalized.slice(end).replace(/^:+/, ""); - const key = `${archivePath}\0${subPath}`; - if (seen.has(key)) continue; - seen.add(key); - candidates.push({ archivePath, subPath }); - } - - return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length); -} - -export class ArchiveReader { - readonly format: ArchiveFormat; - #entries = new Map(); - - constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) { - this.format = format; - for (const entry of entries) { - upsertArchiveEntry(this.#entries, entry); - } - ensureParentDirectories(this.#entries); - } - - getNode(subPath?: string): ArchiveNode | undefined { - const normalizedPath = normalizeArchiveLookupPath(subPath); - if (normalizedPath === undefined) return undefined; - if (normalizedPath === "") { - return { path: "", isDirectory: true, size: 0 }; - } - - const entry = this.#entries.get(normalizedPath); - if (!entry) return undefined; - return { - path: entry.path, - isDirectory: entry.isDirectory, - size: entry.size, - mtimeMs: entry.mtimeMs, - }; - } - - listDirectory(subPath?: string): ArchiveDirectoryEntry[] { - const normalizedPath = normalizeArchiveLookupPath(subPath); - if (normalizedPath === undefined) { - throw new ToolError("Archive path cannot contain '..'"); - } - - if (normalizedPath) { - const entry = this.#entries.get(normalizedPath); - if (!entry) { - throw new ToolError(`Archive path '${normalizedPath}' not found`); - } - if (!entry.isDirectory) { - throw new ToolError(`Archive path '${normalizedPath}' is not a directory`); - } - } - - const prefix = normalizedPath ? `${normalizedPath}/` : ""; - const children = new Map(); - - for (const entry of this.#entries.values()) { - if (normalizedPath) { - if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue; - } - - const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path; - const nextSegment = relativePath.split("/")[0]; - if (!nextSegment) continue; - - const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment; - if (children.has(childPath)) continue; - - const childEntry = this.#entries.get(childPath); - const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/"); - children.set(childPath, { - name: nextSegment, - path: childPath, - isDirectory, - size: isDirectory ? 0 : (childEntry?.size ?? entry.size), - mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs, - }); - } - - return [...children.values()].sort((left, right) => - left.name.toLowerCase().localeCompare(right.name.toLowerCase()), - ); - } - - async readFile(subPath: string): Promise { - const normalizedPath = normalizeArchiveLookupPath(subPath); - if (!normalizedPath) { - throw new ToolError("Archive file path is required"); - } - - const entry = this.#entries.get(normalizedPath); - if (!entry) { - throw new ToolError(`Archive file '${normalizedPath}' not found`); - } - if (entry.isDirectory) { - throw new ToolError(`Archive path '${normalizedPath}' is a directory`); - } - if (!entry.storage) { - throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`); - } - if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) { - throw new ToolError( - `Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`, - ); - } - - const bytes = - entry.storage.type === "tar" - ? await entry.storage.file.bytes() - : await readZipFileBytes(entry.storage, entry.size); - - return { - path: entry.path, - isDirectory: false, - size: entry.size, - mtimeMs: entry.mtimeMs, - bytes, - }; - } -} - -export async function openArchive(filePath: string): Promise { - const format = getArchiveFormatFromPath(filePath); - if (!format) { - throw new ToolError(`Unsupported archive format: ${filePath}`); - } - - if (format === "zip") { - return new ArchiveReader(format, await readZipEntries(filePath)); - } - - const file = Bun.file(filePath); - const archiveSize = file.size; - if (archiveSize > MAX_TAR_ARCHIVE_BYTES) { - throw new ToolError( - `Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`, - ); - } - const entries = await readTarEntries(await file.bytes()); - return new ArchiveReader(format, entries); -} - -export async function listArchiveRoot( - bytes: Uint8Array, - format: ArchiveFormat, - opts: { limit?: number } = {}, -): Promise { - const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-archive-")); - const tempPath = path.join(tempDir, `payload.${format}`); - try { - await Bun.write(tempPath, bytes); - const archive = await openArchive(tempPath); - const entries = archive.listDirectory(""); - const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries; - const lines = formatArchiveEntryLines(limitedEntries); - return lines.length > 0 ? lines.join("\n") : "(empty archive directory)"; - } finally { - await fs.rm(tempDir, { recursive: true, force: true }); - } -} diff --git a/packages/coding-agent/src/tools/fetch.ts b/packages/coding-agent/src/tools/fetch.ts index 42b9d75e9..40314ab0d 100644 --- a/packages/coding-agent/src/tools/fetch.ts +++ b/packages/coding-agent/src/tools/fetch.ts @@ -20,12 +20,12 @@ import { CachedOutputBlock, markFramedBlockComponent } from "../tui/output-block import { webpExclusionForModel } from "../utils/image-loading"; import { formatDimensionNote, resizeImage } from "../utils/image-resize"; import { ensureTool } from "../utils/tools-manager"; +import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "../utils/zip"; import { extractWithParallel, findParallelApiKey, getParallelExtractContent } from "../web/parallel"; import { specialHandlers } from "../web/scrapers"; import type { RenderResult } from "../web/scrapers/types"; import { finalizeOutput, loadPage, looksLikeHtml, MAX_BYTES, MAX_OUTPUT_CHARS } from "../web/scrapers/types"; import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils"; -import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "./archive-reader"; import { applyListLimit } from "./list-limit"; import { formatStyledArtifactReference, type OutputMeta } from "./output-meta"; import { type LineRange, parseLineRanges } from "./path-utils"; diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index d925a5d3e..1aa84fdd9 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -48,8 +48,8 @@ import { webpExclusionForModel, } from "../utils/image-loading"; import { convertFileWithMarkit } from "../utils/markit"; +import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "../utils/zip"; import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree"; -import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader"; import { type ConflictEntry, type ConflictScope, diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index d601026df..51547edf2 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -28,13 +28,8 @@ import { uriHyperlink, } from "../tui"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; +import { type ArchiveReader, type ExtractedArchiveFile, openArchive, parseArchivePathCandidates } from "../utils/zip"; import type { ToolSession } from "."; -import { - type ArchiveReader, - type ExtractedArchiveFile, - openArchive, - parseArchivePathCandidates, -} from "./archive-reader"; import { createFileRecorder, formatResultPath } from "./file-recorder"; import { classifyGroupedLines, formatGroupedFiles, groupLineIndicesByBlank } from "./grouped-file-output"; import { formatMatchLine } from "./match-line-format"; diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index e0cfced7b..ade8a4b46 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -20,8 +20,14 @@ import writeDescription from "../prompts/tools/write.md" with { type: "text" }; import type { ToolSession } from "../sdk"; import { fileHyperlink, framedBlock, renderStatusLine } from "../tui"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; +import { + type ArchiveMemberContent, + archiveFormatFromPath, + parseArchivePathCandidates, + readArchiveEntries, + writeArchive, +} from "../utils/zip"; import { truncateForPrompt } from "./approval"; -import { parseArchivePathCandidates } from "./archive-reader"; import { assertEditableFile } from "./auto-generated-guard"; import { type ConflictEntry, @@ -363,9 +369,10 @@ export class WriteTool implements AgentTool resolvedArchivePath.absolutePath) : resolvedArchivePath.absolutePath; - const lowerPath = finalPath.toLowerCase(); - const isZip = lowerPath.endsWith(".zip"); - const isGzip = lowerPath.endsWith(".tar.gz") || lowerPath.endsWith(".tgz"); + // A realpath swap can land on a name without an archive extension; a + // whole-archive rewrite then defaults to an uncompressed tar, matching the + // previous `isZip`/`isGzip`/else fallthrough. + const format = archiveFormatFromPath(finalPath) ?? "tar"; // Rewrites are whole-archive: write to a temp file and rename so a // crash/disk-full mid-write can't destroy the original archive. const tmpPath = `${finalPath}.tmp-${process.pid}`; @@ -375,66 +382,25 @@ export class WriteTool implements AgentTool = {}; - - if (resolvedArchivePath.exists) { - try { - const bytes = await Bun.file(resolvedArchivePath.absolutePath).bytes(); - const { unzip } = await import("../utils/zip"); - const existing = unzip(new Uint8Array(bytes)); - for (const [entryPath, data] of Object.entries(existing)) { - zipEntries[entryPath.replace(/\\/g, "/")] = data; - } - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } - } - - zipEntries[resolvedArchivePath.archiveSubPath] = new TextEncoder().encode(content); - + const entries = new Map(); + if (resolvedArchivePath.exists) { try { - const { zip } = await import("../utils/zip"); - const zipBuffer = zip(zipEntries); - await Bun.write(tmpPath, zipBuffer); - await fs.rename(tmpPath, finalPath); + const existing = await readArchiveEntries({ bytes: await Bun.file(finalPath).bytes(), format }); + for (const [entryPath, data] of existing) { + entries.set(entryPath, data); + } } catch (error) { - await fs.rm(tmpPath, { force: true }).catch(() => {}); throw new ToolError(error instanceof Error ? error.message : String(error)); } - } else { - const archiveEntries: Record = {}; - if (resolvedArchivePath.exists) { - let archive: Bun.Archive; - try { - archive = new Bun.Archive(await Bun.file(resolvedArchivePath.absolutePath).bytes()); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } + } + entries.set(resolvedArchivePath.archiveSubPath, content); - let files: Map; - try { - files = await archive.files(); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } - - for (const [entryPath, file] of files) { - archiveEntries[entryPath.replace(/\\/g, "/")] = file; - } - } - - archiveEntries[resolvedArchivePath.archiveSubPath] = content; - - try { - // `Bun.Archive.write` never infers compression from the extension; - // request gzip explicitly so `.tar.gz`/`.tgz` stay compressed. - await Bun.Archive.write(tmpPath, archiveEntries, isGzip ? { compress: "gzip" } : undefined); - await fs.rename(tmpPath, finalPath); - } catch (error) { - await fs.rm(tmpPath, { force: true }).catch(() => {}); - throw new ToolError(error instanceof Error ? error.message : String(error)); - } + try { + await writeArchive(tmpPath, format, entries); + await fs.rename(tmpPath, finalPath); + } catch (error) { + await fs.rm(tmpPath, { force: true }).catch(() => {}); + throw new ToolError(error instanceof Error ? error.message : String(error)); } invalidateFsScanAfterWrite(resolvedArchivePath.absolutePath); diff --git a/packages/coding-agent/src/utils/tools-manager.ts b/packages/coding-agent/src/utils/tools-manager.ts index fb4c1b38e..7d6921ec6 100644 --- a/packages/coding-agent/src/utils/tools-manager.ts +++ b/packages/coding-agent/src/utils/tools-manager.ts @@ -2,6 +2,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $which, APP_NAME, getToolsDir, logger, ptree, TempDir } from "@oh-my-pi/pi-utils"; +import { extractArchive } from "./zip"; const TOOLS_DIR = getToolsDir(); const TOOL_DOWNLOAD_TIMEOUT_MS = 120_000; @@ -220,17 +221,7 @@ async function downloadTool(tool: ToolName, signal?: AbortSignal): Promise; +} + +function assertValidRange(start: number, end: number): void { + if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) { + throw new ToolError("Invalid ZIP archive range"); + } +} + +function fileByteSource(filePath: string): ByteSource { + const file = Bun.file(filePath); + const size = file.size; + if (!Number.isSafeInteger(size)) { + throw new ToolError("ZIP archive is too large to read safely"); + } + return { + size, + async read(start, end) { + assertValidRange(start, end); + const bytes = await file.slice(start, end).bytes(); + if (bytes.byteLength !== end - start) { + throw new ToolError("Invalid ZIP archive: truncated data"); + } + return bytes; + }, + }; +} + +function memoryByteSource(buffer: Uint8Array): ByteSource { + return { + size: buffer.byteLength, + async read(start, end) { + assertValidRange(start, end); + if (end > buffer.byteLength) { + throw new ToolError("Invalid ZIP archive: truncated data"); + } + return buffer.subarray(start, end); + }, + }; +} + +interface TarStorage { + type: "tar"; + file: File; +} + +interface ZipStorage { + type: "zip"; + source: ByteSource; + compressedSize: number; + compression: number; + flags: number; + localHeaderOffset: number; +} + +type EntryStorage = TarStorage | ZipStorage; + +interface ArchiveIndexEntry extends ArchiveNode { + storage?: EntryStorage; +} + +function normalizeArchiveLookupPath(rawPath?: string): string | undefined { + if (!rawPath) return ""; + + const parts = rawPath.replace(/\\/g, "/").split("/"); + const normalizedParts: string[] = []; + for (const part of parts) { + if (!part || part === ".") continue; + if (part === "..") return undefined; + normalizedParts.push(part); + } + + return normalizedParts.join("/"); +} + +function normalizeArchiveEntryPath(rawPath: string): string | undefined { + const parts = rawPath.replace(/\\/g, "/").split("/"); + const normalizedParts: string[] = []; + for (const part of parts) { + if (!part || part === ".") continue; + if (part === "..") return undefined; + normalizedParts.push(part); + } + + if (normalizedParts.length === 0) return undefined; + return normalizedParts.join("/"); +} + +function isArchiveDirectoryName(rawPath: string): boolean { + return rawPath.endsWith("/") || rawPath.endsWith("\\"); +} + +function upsertArchiveEntry(map: Map, entry: ArchiveIndexEntry): void { + const existing = map.get(entry.path); + if (!existing) { + map.set(entry.path, entry); + return; + } + + if (existing.isDirectory && !entry.isDirectory) { + map.set(entry.path, entry); + return; + } + + if (!existing.isDirectory && entry.isDirectory) { + return; + } + + map.set(entry.path, { + ...existing, + size: existing.size || entry.size, + mtimeMs: existing.mtimeMs ?? entry.mtimeMs, + storage: existing.storage ?? entry.storage, + }); +} + +function ensureParentDirectories(map: Map): void { + for (const entry of [...map.values()]) { + const parts = entry.path.split("/"); + const stop = parts.length - 1; + for (let index = 1; index <= stop; index++) { + const dirPath = parts.slice(0, index).join("/"); + if (!dirPath || map.has(dirPath)) continue; + map.set(dirPath, { + path: dirPath, + isDirectory: true, + size: 0, + }); + } + } +} + +/** Infer an archive format from a filesystem path's extension. */ +export function archiveFormatFromPath(filePath: string): ArchiveFormat | undefined { + const normalized = filePath.toLowerCase(); + if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz"; + if (normalized.endsWith(".tar")) return "tar"; + if (normalized.endsWith(".zip")) return "zip"; + return undefined; +} + +export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] { + return entries.map(entry => { + if (entry.isDirectory) return `${entry.name}/`; + + const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : ""; + return `${entry.name}${sizeSuffix}`; + }); +} + +const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50; +const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50; +const ZIP64_EOCD_SIGNATURE = 0x06064b50; +const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50; +const ZIP_EOCD_SIGNATURE = 0x06054b50; +const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50; +const ZIP_EOCD_MIN_LENGTH = 22; +const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff; +const ZIP64_EOCD_LOCATOR_LENGTH = 20; +const ZIP_STORED_COMPRESSION = 0; +const ZIP_DEFLATE_COMPRESSION = 8; +const ZIP_UTF8_FLAG = 0x0800; +const ZIP_ENCRYPTED_FLAG = 0x0001; +const ZIP_UINT16_MAX = 0xffff; +const ZIP_UINT32_MAX = 0xffffffff; +const ZIP_UINT32_RANGE = 0x100000000; + +interface ZipCentralDirectoryInfo { + entries: number; + offset: number; + size: number; +} + +interface Zip64EntryValues { + compressedSize: number; + uncompressedSize: number; + localHeaderOffset: number; + diskStart: number; +} + +interface Zip64EntryPlaceholders { + compressedSize: boolean; + uncompressedSize: boolean; + localHeaderOffset: boolean; + diskStart: boolean; +} + +function readUInt16LE(bytes: Uint8Array, offset: number): number { + return bytes[offset]! | (bytes[offset + 1]! << 8); +} + +function readUInt32LE(bytes: Uint8Array, offset: number): number { + return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0; +} + +function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean { + if (bytes.byteLength < offset + value.length) return false; + for (let index = 0; index < value.length; index++) { + if (bytes[offset + index] !== value.charCodeAt(index)) return false; + } + return true; +} + +export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined { + if (bytes.byteLength >= 4) { + const signature = readUInt32LE(bytes, 0); + if ( + signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE || + signature === ZIP_EOCD_SIGNATURE || + signature === ZIP_DATA_DESCRIPTOR_SIGNATURE + ) { + return "zip"; + } + } + + if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) { + return "tar.gz"; + } + + if (bytesMatchAscii(bytes, 257, "ustar")) { + return "tar"; + } + + return undefined; +} + +function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number { + const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE; + if (!Number.isSafeInteger(value)) { + throw new ToolError("ZIP archive uses offsets or sizes too large to read safely"); + } + return value; +} + +function findEndOfCentralDirectory(tail: Uint8Array): number { + for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) { + if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue; + const commentLength = readUInt16LE(tail, offset + 20); + if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset; + } + + throw new ToolError("Invalid ZIP archive: missing end of central directory"); +} + +async function readZip64CentralDirectoryInfo( + source: ByteSource, + tail: Uint8Array, + tailStart: number, + eocdOffset: number, +): Promise { + const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH; + if (locatorOffset < 0) return undefined; + + const locator = + locatorOffset >= tailStart + ? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH) + : await source.read(locatorOffset, eocdOffset); + if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined; + + const zip64EocdDisk = readUInt32LE(locator, 4); + const zip64EocdOffset = readUInt64LEAsNumber(locator, 8); + const totalDisks = readUInt32LE(locator, 16); + if (zip64EocdDisk !== 0 || totalDisks > 1) { + throw new ToolError("Multi-disk ZIP archives are not supported"); + } + + const record = await source.read(zip64EocdOffset, zip64EocdOffset + 56); + if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) { + throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory"); + } + if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) { + throw new ToolError("Multi-disk ZIP archives are not supported"); + } + + return { + entries: readUInt64LEAsNumber(record, 32), + size: readUInt64LEAsNumber(record, 40), + offset: readUInt64LEAsNumber(record, 48), + }; +} + +async function readZipCentralDirectoryInfo(source: ByteSource): Promise { + const fileSize = source.size; + if (fileSize < ZIP_EOCD_MIN_LENGTH) { + throw new ToolError("Invalid ZIP archive: missing end of central directory"); + } + + const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH); + const tailStart = fileSize - tailLength; + const tail = await source.read(tailStart, fileSize); + const eocdIndex = findEndOfCentralDirectory(tail); + const eocdOffset = tailStart + eocdIndex; + + if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) { + throw new ToolError("Multi-disk ZIP archives are not supported"); + } + + let entries = readUInt16LE(tail, eocdIndex + 10); + let size = readUInt32LE(tail, eocdIndex + 12); + let offset = readUInt32LE(tail, eocdIndex + 16); + const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX; + const zip64Info = await readZip64CentralDirectoryInfo(source, tail, tailStart, eocdOffset); + if (zip64Info) { + ({ entries, size, offset } = zip64Info); + } else if (needsZip64) { + throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata"); + } + + if (offset + size > fileSize) { + throw new ToolError("Invalid ZIP archive: central directory exceeds file size"); + } + + return { entries, offset, size }; +} + +function readZip64EntryValues( + extra: Uint8Array, + placeholders: Zip64EntryPlaceholders, + current: Zip64EntryValues, +): Zip64EntryValues { + if ( + !placeholders.compressedSize && + !placeholders.uncompressedSize && + !placeholders.localHeaderOffset && + !placeholders.diskStart + ) { + return current; + } + + let offset = 0; + while (offset + 4 <= extra.byteLength) { + const headerId = readUInt16LE(extra, offset); + const dataSize = readUInt16LE(extra, offset + 2); + const dataStart = offset + 4; + const dataEnd = dataStart + dataSize; + if (dataEnd > extra.byteLength) { + throw new ToolError("Invalid ZIP archive: malformed extra field"); + } + + if (headerId === 0x0001) { + let cursor = dataStart; + let uncompressedSize = current.uncompressedSize; + let compressedSize = current.compressedSize; + let localHeaderOffset = current.localHeaderOffset; + let diskStart = current.diskStart; + + if (placeholders.uncompressedSize) { + if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); + uncompressedSize = readUInt64LEAsNumber(extra, cursor); + cursor += 8; + } + if (placeholders.compressedSize) { + if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); + compressedSize = readUInt64LEAsNumber(extra, cursor); + cursor += 8; + } + if (placeholders.localHeaderOffset) { + if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); + localHeaderOffset = readUInt64LEAsNumber(extra, cursor); + cursor += 8; + } + if (placeholders.diskStart) { + if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field"); + diskStart = readUInt32LE(extra, cursor); + } + + return { compressedSize, uncompressedSize, localHeaderOffset, diskStart }; + } + + offset = dataEnd; + } + + throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field"); +} + +function parseZipCentralDirectory( + source: ByteSource, + centralDirectory: Uint8Array, + expectedEntries: number, +): ArchiveIndexEntry[] { + const entries: ArchiveIndexEntry[] = []; + let offset = 0; + + for (let index = 0; index < expectedEntries; index++) { + if (offset + 46 > centralDirectory.byteLength) { + throw new ToolError("Invalid ZIP archive: truncated central directory"); + } + if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) { + throw new ToolError("Invalid ZIP archive: malformed central directory"); + } + + const flags = readUInt16LE(centralDirectory, offset + 8); + const compression = readUInt16LE(centralDirectory, offset + 10); + const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20); + const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24); + const fileNameLength = readUInt16LE(centralDirectory, offset + 28); + const extraLength = readUInt16LE(centralDirectory, offset + 30); + const commentLength = readUInt16LE(centralDirectory, offset + 32); + const diskStartRaw = readUInt16LE(centralDirectory, offset + 34); + const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42); + const nameStart = offset + 46; + const extraStart = nameStart + fileNameLength; + const entryEnd = extraStart + extraLength + commentLength; + if (entryEnd > centralDirectory.byteLength) { + throw new ToolError("Invalid ZIP archive: truncated central directory entry"); + } + + const useLegacyEncoding = (flags & ZIP_UTF8_FLAG) === 0; + const rawPath = (useLegacyEncoding ? LEGACY_NAME_DECODER : UTF8_DECODER).decode( + centralDirectory.subarray(nameStart, extraStart), + ); + const normalizedPath = normalizeArchiveEntryPath(rawPath); + if (normalizedPath) { + const values = readZip64EntryValues( + centralDirectory.subarray(extraStart, extraStart + extraLength), + { + compressedSize: compressedSizeRaw === ZIP_UINT32_MAX, + uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX, + localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX, + diskStart: diskStartRaw === ZIP_UINT16_MAX, + }, + { + compressedSize: compressedSizeRaw, + uncompressedSize: uncompressedSizeRaw, + localHeaderOffset: localHeaderOffsetRaw, + diskStart: diskStartRaw, + }, + ); + if (values.diskStart !== 0) { + throw new ToolError("Multi-disk ZIP archives are not supported"); + } + + const isDirectory = isArchiveDirectoryName(rawPath); + entries.push({ + path: normalizedPath, + isDirectory, + size: isDirectory ? 0 : values.uncompressedSize, + storage: isDirectory + ? undefined + : { + type: "zip", + source, + compressedSize: values.compressedSize, + compression, + flags, + localHeaderOffset: values.localHeaderOffset, + }, + }); + } + + offset = entryEnd; + } + + return entries; +} + +async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise { + if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) { + throw new ToolError("Encrypted ZIP entries are not supported"); + } + + const localHeader = await storage.source.read(storage.localHeaderOffset, storage.localHeaderOffset + 30); + if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) { + throw new ToolError("Invalid ZIP archive: malformed local file header"); + } + + const fileNameLength = readUInt16LE(localHeader, 26); + const extraLength = readUInt16LE(localHeader, 28); + const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength; + const compressedBytes = await storage.source.read(dataStart, dataStart + storage.compressedSize); + + if (storage.compression === ZIP_STORED_COMPRESSION) { + return compressedBytes; + } + if (storage.compression !== ZIP_DEFLATE_COMPRESSION) { + throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`); + } + + try { + return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize)); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } +} + +async function readTarEntries(bytes: Uint8Array): Promise { + let archive: Bun.Archive; + try { + archive = new Bun.Archive(bytes); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } + + let files: Map; + try { + files = await archive.files(); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } + + const entries: ArchiveIndexEntry[] = []; + for (const [rawPath, file] of files) { + const normalizedPath = normalizeArchiveEntryPath(rawPath); + if (!normalizedPath) continue; + const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined; + entries.push({ + path: normalizedPath, + isDirectory: false, + size: file.size, + mtimeMs, + storage: { type: "tar", file }, + }); + } + + return entries; +} + +async function readZipEntries(source: ByteSource): Promise { + const directoryInfo = await readZipCentralDirectoryInfo(source); + const centralDirectory = await source.read(directoryInfo.offset, directoryInfo.offset + directoryInfo.size); + return parseZipCentralDirectory(source, centralDirectory, directoryInfo.entries); +} + +/** + * Split an `archive.ext:inner/path` reference into every plausible + * `{ archivePath, subPath }` pair, longest archive prefix first. A path may + * contain more than one archive extension, so each candidate is a guess at + * where the archive ends and the member portion begins. + */ +export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] { + const normalized = filePath.replace(/\\/g, "/"); + const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi; + const seen = new Set(); + const candidates: ArchivePathCandidate[] = []; + + let match: RegExpExecArray | null; + while (true) { + match = pattern.exec(normalized); + if (match === null) { + break; + } + const end = match.index + match[0].length; + const archivePath = filePath.slice(0, end); + const subPath = normalized.slice(end).replace(/^:+/, ""); + const key = `${archivePath}\0${subPath}`; + if (seen.has(key)) continue; + seen.add(key); + candidates.push({ archivePath, subPath }); + } + + return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length); +} + +/** + * An indexed, read-only view over a single archive. ZIP archives are indexed + * from the central directory and members are inflated on demand; tar archives + * are fully materialized by `Bun.Archive` up front. + */ +export class ArchiveReader { + readonly format: ArchiveFormat; + #entries = new Map(); + + constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) { + this.format = format; + for (const entry of entries) { + upsertArchiveEntry(this.#entries, entry); + } + ensureParentDirectories(this.#entries); + } + + getNode(subPath?: string): ArchiveNode | undefined { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (normalizedPath === undefined) return undefined; + if (normalizedPath === "") { + return { path: "", isDirectory: true, size: 0 }; + } + + const entry = this.#entries.get(normalizedPath); + if (!entry) return undefined; + return { + path: entry.path, + isDirectory: entry.isDirectory, + size: entry.size, + mtimeMs: entry.mtimeMs, + }; + } + + listDirectory(subPath?: string): ArchiveDirectoryEntry[] { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (normalizedPath === undefined) { + throw new ToolError("Archive path cannot contain '..'"); + } + + if (normalizedPath) { + const entry = this.#entries.get(normalizedPath); + if (!entry) { + throw new ToolError(`Archive path '${normalizedPath}' not found`); + } + if (!entry.isDirectory) { + throw new ToolError(`Archive path '${normalizedPath}' is not a directory`); + } + } + + const prefix = normalizedPath ? `${normalizedPath}/` : ""; + const children = new Map(); + + for (const entry of this.#entries.values()) { + if (normalizedPath) { + if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue; + } + + const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path; + const nextSegment = relativePath.split("/")[0]; + if (!nextSegment) continue; + + const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment; + if (children.has(childPath)) continue; + + const childEntry = this.#entries.get(childPath); + const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/"); + children.set(childPath, { + name: nextSegment, + path: childPath, + isDirectory, + size: isDirectory ? 0 : (childEntry?.size ?? entry.size), + mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs, + }); + } + + return [...children.values()].sort((left, right) => + left.name.toLowerCase().localeCompare(right.name.toLowerCase()), + ); + } + + async readFile(subPath: string): Promise { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (!normalizedPath) { + throw new ToolError("Archive file path is required"); + } + + const entry = this.#entries.get(normalizedPath); + if (!entry) { + throw new ToolError(`Archive file '${normalizedPath}' not found`); + } + if (entry.isDirectory) { + throw new ToolError(`Archive path '${normalizedPath}' is a directory`); + } + if (!entry.storage) { + throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`); + } + if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) { + throw new ToolError( + `Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`, + ); + } + + const bytes = + entry.storage.type === "tar" + ? await entry.storage.file.bytes() + : await readZipFileBytes(entry.storage, entry.size); + + return { + path: entry.path, + isDirectory: false, + size: entry.size, + mtimeMs: entry.mtimeMs, + bytes, + }; + } +} + +/** + * Open an archive for reading. ZIP archives opened from a path are indexed + * lazily via ranged central-directory reads (members inflate on demand); tar + * archives and in-memory ZIPs are read from a single buffer. + */ +export async function openArchive(source: ArchiveSource): Promise { + if (typeof source === "string") { + const format = archiveFormatFromPath(source); + if (!format) { + throw new ToolError(`Unsupported archive format: ${source}`); + } + if (format === "zip") { + return new ArchiveReader(format, await readZipEntries(fileByteSource(source))); + } + + const file = Bun.file(source); + const archiveSize = file.size; + if (archiveSize > MAX_TAR_ARCHIVE_BYTES) { + throw new ToolError( + `Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`, + ); + } + return new ArchiveReader(format, await readTarEntries(await file.bytes())); + } + + const { bytes, format } = source; + if (format === "zip") { + return new ArchiveReader(format, await readZipEntries(memoryByteSource(bytes))); + } + if (bytes.byteLength > MAX_TAR_ARCHIVE_BYTES) { + throw new ToolError( + `Archive is too large to read in memory (${formatBytes(bytes.byteLength)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`, + ); + } + return new ArchiveReader(format, await readTarEntries(bytes)); +} + +/** Render the top-level entries of an in-memory archive as one line each. */ +export async function listArchiveRoot( + bytes: Uint8Array, + format: ArchiveFormat, + opts: { limit?: number } = {}, +): Promise { + const archive = await openArchive({ bytes, format }); + const entries = archive.listDirectory(""); + const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries; + const lines = formatArchiveEntryLines(limitedEntries); + return lines.length > 0 ? lines.join("\n") : "(empty archive directory)"; +} + +async function resolveArchiveBytes(source: ArchiveSource): Promise<{ bytes: Uint8Array; format: ArchiveFormat }> { + if (typeof source !== "string") return source; + const format = archiveFormatFromPath(source); + if (!format) { + throw new ToolError(`Unsupported archive format: ${source}`); + } + return { bytes: await Bun.file(source).bytes(), format }; +} + +async function memberToBytes(content: ArchiveMemberContent): Promise { + if (typeof content === "string") return ENCODER.encode(content); + if (content instanceof Uint8Array) return content; + return new Uint8Array(await content.arrayBuffer()); +} + +/** + * Fully materialize every file member into a `path → content` map: ZIP members + * are inflated via fflate, tar members are returned as lazy `File`s. Use this + * when you need every entry (rewrite, extract); for browsing or single-member + * reads prefer `openArchive`, which is lazy for ZIP. + */ +export async function readArchiveEntries(source: ArchiveSource): Promise> { + const { bytes, format } = await resolveArchiveBytes(source); + const entries = new Map(); + if (format === "zip") { + const unzipped = unzipSync(bytes); + for (const name in unzipped) { + entries.set(name.replace(/\\/g, "/"), unzipped[name]!); + } + return entries; + } + const files = await new Bun.Archive(bytes).files(); + for (const [name, file] of files) { + entries.set(name.replace(/\\/g, "/"), file); + } + return entries; +} + +/** + * Serialize `entries` into an archive of `format` and write it to `destPath`. + * ZIP is built with fflate, tar / tar.gz with `Bun.Archive` (gzip for tar.gz). + * String members are encoded as UTF-8. + */ +export async function writeArchive( + destPath: string, + format: ArchiveFormat, + entries: Iterable, +): Promise { + if (format === "zip") { + const record: Record = {}; + for (const [name, content] of entries) { + record[name.replace(/\\/g, "/")] = await memberToBytes(content); + } + await Bun.write(destPath, zipSync(record)); + return; + } + + const record: Record = {}; + for (const [name, content] of entries) { + record[name.replace(/\\/g, "/")] = content; + } + await Bun.Archive.write(destPath, record, format === "tar.gz" ? { compress: "gzip" } : undefined); +} + +/** + * Extract every file member to `destDir`, creating parent directories as + * needed. Entries that would escape `destDir` (via `..` or an absolute path) + * are rejected. Returns the number of files written. + */ +export async function extractArchive(source: ArchiveSource, destDir: string): Promise { + const extractRoot = path.resolve(destDir); + const entries = await readArchiveEntries(source); + let count = 0; + for (const [name, content] of entries) { + if (name.endsWith("/")) continue; + const outputPath = path.resolve(extractRoot, name); + if (!outputPath.startsWith(extractRoot + path.sep)) { + throw new ToolError(`Archive entry escapes extraction dir: ${name}`); + } + await Bun.write(outputPath, content); + count++; + } + return count; }