feat(coding-agent): consolidated and optimize archive handling

- Centralized archive operations into a new `utils/zip.ts` module with unified support for ZIP, tar, and tar.gz formats.
- Optimized ZIP reading using lazy, ranged central-directory access and implemented ZIP64 support for large files.
- Hardened archive extraction with directory traversal protection and configured memory limits for loading and extraction.
- Refactored tool-specific logic to utilize the new centralized utility and deleted the redundant `archive-reader.ts`.
This commit is contained in:
can1357
2026-06-18 19:20:52 +02:00
parent 38d1d3a2f5
commit 021d82ac1f
11 changed files with 911 additions and 815 deletions
+1 -1
View File
@@ -7,7 +7,7 @@
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/read.md`
- Key collaborators:
- `packages/coding-agent/src/tools/path-utils.ts` — split `path` from trailing selectors; normalize local paths.
- `packages/coding-agent/src/tools/archive-reader.ts` — detect `archive.ext:inner/path`, index archives, list/read entries.
- `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: detect `archive.ext:inner/path`, index archives, list/read entries.
- `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite targets, parse selectors, render tables.
- `packages/coding-agent/src/tools/fetch.ts` — URL parsing, fetch/render pipeline, URL cache/artifacts.
- `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`.
+2 -2
View File
@@ -6,7 +6,7 @@
- Entry: `packages/coding-agent/src/tools/write.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/write.md`
- Key collaborators:
- `packages/coding-agent/src/tools/archive-reader.ts` — parse `archive.ext:entry` selectors.
- `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: parse `archive.ext:entry` selectors and rewrite the archive whole.
- `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite paths and perform row insert/update/delete.
- `packages/coding-agent/src/tools/conflict-detect.ts` — parse `conflict://` URIs and splice recorded merge-conflict regions.
- `packages/coding-agent/src/lsp/index.ts` — format-on-write and diagnostics writethrough.
@@ -56,7 +56,7 @@ Single-shot result.
1. `WriteTool.execute()` in `packages/coding-agent/src/tools/write.ts` strips pasted `[PATH#HASH]` headers and `LINE:` hashline prefixes from `content` when the session is in hashline display mode.
2. If `path` is an internal URL whose handler exposes `write`, the tool delegates directly to `handler.write(...)` and returns.
3. `conflict://...` paths are handled next by the merge-conflict resolver. Scope reads such as `conflict://<id>/ours` are rejected as read-only; writable conflict URIs must omit the scope.
4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/tools/archive-reader.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet.
4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/utils/zip.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet.
5. Archive writes call `enforcePlanModeWrite(..., { op: exists ? "update" : "create" })`, then `#writeArchiveEntry()`.
- The parent directory of the archive file is created with `fs.mkdir(..., { recursive: true })`.
- `.zip` archives are read with `fflate.unzipSync()`, the target entry is replaced in an in-memory map, and the archive is rewritten with `fflate.zipSync()` + `Bun.write()`.
+3 -1
View File
@@ -1,9 +1,11 @@
# Changelog
## [Unreleased]
### Changed
- Refactored internal archive handling into a unified `src/utils/zip.ts` module
- Centralized all `fflate` (ZIP) and `Bun.Archive` (tar/tar.gz) operations into `zip.ts`
- Optimized archive reading by using lazy, ranged central-directory access for ZIP files
- Updated internal image processing to no longer include metadata text for fetched images
- Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob
- Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved.
@@ -7,6 +7,7 @@ import * as fs from "node:fs/promises";
import * as path from "node:path";
import type { WorkProfile } from "@oh-my-pi/pi-natives";
import { APP_NAME, getLogPath, getLogsDir, getReportsDir, isEnoent } from "@oh-my-pi/pi-utils";
import { writeArchive } from "../utils/zip";
import type { CpuProfile, HeapSnapshot } from "./profiler";
import { collectSystemInfo, sanitizeEnv } from "./system-info";
@@ -165,7 +166,7 @@ export async function createReportBundle(options: ReportBundleOptions): Promise<
}
// Write archive
await Bun.Archive.write(outputPath, data, { compress: "gzip" });
await writeArchive(outputPath, "tar.gz", Object.entries(data));
return { path: outputPath, files };
}
@@ -1,721 +0,0 @@
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { bytesToText, inflateRaw } from "../utils/zip";
import { formatBytes } from "./render-utils";
import { ToolError } from "./tool-errors";
/**
* Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into
* memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is
* exempt: it is read via ranged central-directory access.
*/
const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024;
/**
* Cap on a single archive member's declared (uncompressed) size. The declared
* size is attacker-controlled metadata — a crafted ZIP entry can claim
* multi-GB sizes that would be allocated up front before any data inflates.
*/
const MAX_ARCHIVE_MEMBER_BYTES = 64 * 1024 * 1024;
export type ArchiveFormat = "zip" | "tar" | "tar.gz";
export interface ArchivePathCandidate {
archivePath: string;
subPath: string;
}
export interface ArchiveNode {
path: string;
isDirectory: boolean;
size: number;
mtimeMs?: number;
}
export interface ArchiveDirectoryEntry extends ArchiveNode {
name: string;
}
export interface ExtractedArchiveFile extends ArchiveNode {
bytes: Uint8Array;
}
interface TarStorage {
type: "tar";
file: File;
}
interface ZipStorage {
type: "zip";
archivePath: string;
compressedSize: number;
compression: number;
flags: number;
localHeaderOffset: number;
}
type EntryStorage = TarStorage | ZipStorage;
interface ArchiveIndexEntry extends ArchiveNode {
storage?: EntryStorage;
}
function normalizeArchiveLookupPath(rawPath?: string): string | undefined {
if (!rawPath) return "";
const parts = rawPath.replace(/\\/g, "/").split("/");
const normalizedParts: string[] = [];
for (const part of parts) {
if (!part || part === ".") continue;
if (part === "..") return undefined;
normalizedParts.push(part);
}
return normalizedParts.join("/");
}
function normalizeArchiveEntryPath(rawPath: string): string | undefined {
const parts = rawPath.replace(/\\/g, "/").split("/");
const normalizedParts: string[] = [];
for (const part of parts) {
if (!part || part === ".") continue;
if (part === "..") return undefined;
normalizedParts.push(part);
}
if (normalizedParts.length === 0) return undefined;
return normalizedParts.join("/");
}
function isArchiveDirectoryName(rawPath: string): boolean {
return rawPath.endsWith("/") || rawPath.endsWith("\\");
}
function upsertArchiveEntry(map: Map<string, ArchiveIndexEntry>, entry: ArchiveIndexEntry): void {
const existing = map.get(entry.path);
if (!existing) {
map.set(entry.path, entry);
return;
}
if (existing.isDirectory && !entry.isDirectory) {
map.set(entry.path, entry);
return;
}
if (!existing.isDirectory && entry.isDirectory) {
return;
}
map.set(entry.path, {
...existing,
size: existing.size || entry.size,
mtimeMs: existing.mtimeMs ?? entry.mtimeMs,
storage: existing.storage ?? entry.storage,
});
}
function ensureParentDirectories(map: Map<string, ArchiveIndexEntry>): void {
for (const entry of [...map.values()]) {
const parts = entry.path.split("/");
const stop = parts.length - 1;
for (let index = 1; index <= stop; index++) {
const dirPath = parts.slice(0, index).join("/");
if (!dirPath || map.has(dirPath)) continue;
map.set(dirPath, {
path: dirPath,
isDirectory: true,
size: 0,
});
}
}
}
function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined {
const normalized = filePath.toLowerCase();
if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz";
if (normalized.endsWith(".tar")) return "tar";
if (normalized.endsWith(".zip")) return "zip";
return undefined;
}
export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] {
return entries.map(entry => {
if (entry.isDirectory) return `${entry.name}/`;
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
return `${entry.name}${sizeSuffix}`;
});
}
const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50;
const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50;
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50;
const ZIP_EOCD_SIGNATURE = 0x06054b50;
const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50;
const ZIP_EOCD_MIN_LENGTH = 22;
const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff;
const ZIP64_EOCD_LOCATOR_LENGTH = 20;
const ZIP_STORED_COMPRESSION = 0;
const ZIP_DEFLATE_COMPRESSION = 8;
const ZIP_UTF8_FLAG = 0x0800;
const ZIP_ENCRYPTED_FLAG = 0x0001;
const ZIP_UINT16_MAX = 0xffff;
const ZIP_UINT32_MAX = 0xffffffff;
const ZIP_UINT32_RANGE = 0x100000000;
interface ZipCentralDirectoryInfo {
entries: number;
offset: number;
size: number;
}
interface Zip64EntryValues {
compressedSize: number;
uncompressedSize: number;
localHeaderOffset: number;
diskStart: number;
}
interface Zip64EntryPlaceholders {
compressedSize: boolean;
uncompressedSize: boolean;
localHeaderOffset: boolean;
diskStart: boolean;
}
function readUInt16LE(bytes: Uint8Array, offset: number): number {
return bytes[offset]! | (bytes[offset + 1]! << 8);
}
function readUInt32LE(bytes: Uint8Array, offset: number): number {
return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0;
}
function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean {
if (bytes.byteLength < offset + value.length) return false;
for (let index = 0; index < value.length; index++) {
if (bytes[offset + index] !== value.charCodeAt(index)) return false;
}
return true;
}
export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined {
if (bytes.byteLength >= 4) {
const signature = readUInt32LE(bytes, 0);
if (
signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE ||
signature === ZIP_EOCD_SIGNATURE ||
signature === ZIP_DATA_DESCRIPTOR_SIGNATURE
) {
return "zip";
}
}
if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) {
return "tar.gz";
}
if (bytesMatchAscii(bytes, 257, "ustar")) {
return "tar";
}
return undefined;
}
function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number {
const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE;
if (!Number.isSafeInteger(value)) {
throw new ToolError("ZIP archive uses offsets or sizes too large to read safely");
}
return value;
}
async function readZipRange(filePath: string, start: number, end: number): Promise<Uint8Array> {
if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) {
throw new ToolError("Invalid ZIP archive range");
}
const bytes = await Bun.file(filePath).slice(start, end).bytes();
if (bytes.byteLength !== end - start) {
throw new ToolError("Invalid ZIP archive: truncated data");
}
return bytes;
}
function findEndOfCentralDirectory(tail: Uint8Array): number {
for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) {
if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue;
const commentLength = readUInt16LE(tail, offset + 20);
if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset;
}
throw new ToolError("Invalid ZIP archive: missing end of central directory");
}
async function readZip64CentralDirectoryInfo(
filePath: string,
tail: Uint8Array,
tailStart: number,
eocdOffset: number,
): Promise<ZipCentralDirectoryInfo | undefined> {
const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH;
if (locatorOffset < 0) return undefined;
const locator =
locatorOffset >= tailStart
? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH)
: await readZipRange(filePath, locatorOffset, eocdOffset);
if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined;
const zip64EocdDisk = readUInt32LE(locator, 4);
const zip64EocdOffset = readUInt64LEAsNumber(locator, 8);
const totalDisks = readUInt32LE(locator, 16);
if (zip64EocdDisk !== 0 || totalDisks > 1) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
const record = await readZipRange(filePath, zip64EocdOffset, zip64EocdOffset + 56);
if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory");
}
if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
return {
entries: readUInt64LEAsNumber(record, 32),
size: readUInt64LEAsNumber(record, 40),
offset: readUInt64LEAsNumber(record, 48),
};
}
async function readZipCentralDirectoryInfo(filePath: string, fileSize: number): Promise<ZipCentralDirectoryInfo> {
if (fileSize < ZIP_EOCD_MIN_LENGTH) {
throw new ToolError("Invalid ZIP archive: missing end of central directory");
}
const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH);
const tailStart = fileSize - tailLength;
const tail = await readZipRange(filePath, tailStart, fileSize);
const eocdIndex = findEndOfCentralDirectory(tail);
const eocdOffset = tailStart + eocdIndex;
if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
let entries = readUInt16LE(tail, eocdIndex + 10);
let size = readUInt32LE(tail, eocdIndex + 12);
let offset = readUInt32LE(tail, eocdIndex + 16);
const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX;
const zip64Info = await readZip64CentralDirectoryInfo(filePath, tail, tailStart, eocdOffset);
if (zip64Info) {
({ entries, size, offset } = zip64Info);
} else if (needsZip64) {
throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata");
}
if (offset + size > fileSize) {
throw new ToolError("Invalid ZIP archive: central directory exceeds file size");
}
return { entries, offset, size };
}
function readZip64EntryValues(
extra: Uint8Array,
placeholders: Zip64EntryPlaceholders,
current: Zip64EntryValues,
): Zip64EntryValues {
if (
!placeholders.compressedSize &&
!placeholders.uncompressedSize &&
!placeholders.localHeaderOffset &&
!placeholders.diskStart
) {
return current;
}
let offset = 0;
while (offset + 4 <= extra.byteLength) {
const headerId = readUInt16LE(extra, offset);
const dataSize = readUInt16LE(extra, offset + 2);
const dataStart = offset + 4;
const dataEnd = dataStart + dataSize;
if (dataEnd > extra.byteLength) {
throw new ToolError("Invalid ZIP archive: malformed extra field");
}
if (headerId === 0x0001) {
let cursor = dataStart;
let uncompressedSize = current.uncompressedSize;
let compressedSize = current.compressedSize;
let localHeaderOffset = current.localHeaderOffset;
let diskStart = current.diskStart;
if (placeholders.uncompressedSize) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
uncompressedSize = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.compressedSize) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
compressedSize = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.localHeaderOffset) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
localHeaderOffset = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.diskStart) {
if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
diskStart = readUInt32LE(extra, cursor);
}
return { compressedSize, uncompressedSize, localHeaderOffset, diskStart };
}
offset = dataEnd;
}
throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field");
}
function parseZipCentralDirectory(
filePath: string,
centralDirectory: Uint8Array,
expectedEntries: number,
): ArchiveIndexEntry[] {
const entries: ArchiveIndexEntry[] = [];
let offset = 0;
for (let index = 0; index < expectedEntries; index++) {
if (offset + 46 > centralDirectory.byteLength) {
throw new ToolError("Invalid ZIP archive: truncated central directory");
}
if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: malformed central directory");
}
const flags = readUInt16LE(centralDirectory, offset + 8);
const compression = readUInt16LE(centralDirectory, offset + 10);
const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20);
const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24);
const fileNameLength = readUInt16LE(centralDirectory, offset + 28);
const extraLength = readUInt16LE(centralDirectory, offset + 30);
const commentLength = readUInt16LE(centralDirectory, offset + 32);
const diskStartRaw = readUInt16LE(centralDirectory, offset + 34);
const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42);
const nameStart = offset + 46;
const extraStart = nameStart + fileNameLength;
const entryEnd = extraStart + extraLength + commentLength;
if (entryEnd > centralDirectory.byteLength) {
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
}
const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
const normalizedPath = normalizeArchiveEntryPath(rawPath);
if (normalizedPath) {
const values = readZip64EntryValues(
centralDirectory.subarray(extraStart, extraStart + extraLength),
{
compressedSize: compressedSizeRaw === ZIP_UINT32_MAX,
uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX,
localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX,
diskStart: diskStartRaw === ZIP_UINT16_MAX,
},
{
compressedSize: compressedSizeRaw,
uncompressedSize: uncompressedSizeRaw,
localHeaderOffset: localHeaderOffsetRaw,
diskStart: diskStartRaw,
},
);
if (values.diskStart !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
const isDirectory = isArchiveDirectoryName(rawPath);
entries.push({
path: normalizedPath,
isDirectory,
size: isDirectory ? 0 : values.uncompressedSize,
storage: isDirectory
? undefined
: {
type: "zip",
archivePath: filePath,
compressedSize: values.compressedSize,
compression,
flags,
localHeaderOffset: values.localHeaderOffset,
},
});
}
offset = entryEnd;
}
return entries;
}
async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise<Uint8Array> {
if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) {
throw new ToolError("Encrypted ZIP entries are not supported");
}
const localHeader = await readZipRange(
storage.archivePath,
storage.localHeaderOffset,
storage.localHeaderOffset + 30,
);
if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: malformed local file header");
}
const fileNameLength = readUInt16LE(localHeader, 26);
const extraLength = readUInt16LE(localHeader, 28);
const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength;
const compressedBytes = await readZipRange(storage.archivePath, dataStart, dataStart + storage.compressedSize);
if (storage.compression === ZIP_STORED_COMPRESSION) {
return compressedBytes;
}
if (storage.compression !== ZIP_DEFLATE_COMPRESSION) {
throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`);
}
try {
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
}
async function readTarEntries(bytes: Uint8Array): Promise<ArchiveIndexEntry[]> {
let archive: Bun.Archive;
try {
archive = new Bun.Archive(bytes);
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
let files: Map<string, File>;
try {
files = await archive.files();
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
const entries: ArchiveIndexEntry[] = [];
for (const [rawPath, file] of files) {
const normalizedPath = normalizeArchiveEntryPath(rawPath);
if (!normalizedPath) continue;
const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined;
entries.push({
path: normalizedPath,
isDirectory: false,
size: file.size,
mtimeMs,
storage: { type: "tar", file },
});
}
return entries;
}
async function readZipEntries(filePath: string): Promise<ArchiveIndexEntry[]> {
const fileSize = Bun.file(filePath).size;
if (!Number.isSafeInteger(fileSize)) {
throw new ToolError("ZIP archive is too large to read safely");
}
const directoryInfo = await readZipCentralDirectoryInfo(filePath, fileSize);
const centralDirectory = await readZipRange(
filePath,
directoryInfo.offset,
directoryInfo.offset + directoryInfo.size,
);
return parseZipCentralDirectory(filePath, centralDirectory, directoryInfo.entries);
}
export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] {
const normalized = filePath.replace(/\\/g, "/");
const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi;
const seen = new Set<string>();
const candidates: ArchivePathCandidate[] = [];
let match: RegExpExecArray | null;
while (true) {
match = pattern.exec(normalized);
if (match === null) {
break;
}
const end = match.index + match[0].length;
const archivePath = filePath.slice(0, end);
const subPath = normalized.slice(end).replace(/^:+/, "");
const key = `${archivePath}\0${subPath}`;
if (seen.has(key)) continue;
seen.add(key);
candidates.push({ archivePath, subPath });
}
return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length);
}
export class ArchiveReader {
readonly format: ArchiveFormat;
#entries = new Map<string, ArchiveIndexEntry>();
constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) {
this.format = format;
for (const entry of entries) {
upsertArchiveEntry(this.#entries, entry);
}
ensureParentDirectories(this.#entries);
}
getNode(subPath?: string): ArchiveNode | undefined {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (normalizedPath === undefined) return undefined;
if (normalizedPath === "") {
return { path: "", isDirectory: true, size: 0 };
}
const entry = this.#entries.get(normalizedPath);
if (!entry) return undefined;
return {
path: entry.path,
isDirectory: entry.isDirectory,
size: entry.size,
mtimeMs: entry.mtimeMs,
};
}
listDirectory(subPath?: string): ArchiveDirectoryEntry[] {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (normalizedPath === undefined) {
throw new ToolError("Archive path cannot contain '..'");
}
if (normalizedPath) {
const entry = this.#entries.get(normalizedPath);
if (!entry) {
throw new ToolError(`Archive path '${normalizedPath}' not found`);
}
if (!entry.isDirectory) {
throw new ToolError(`Archive path '${normalizedPath}' is not a directory`);
}
}
const prefix = normalizedPath ? `${normalizedPath}/` : "";
const children = new Map<string, ArchiveDirectoryEntry>();
for (const entry of this.#entries.values()) {
if (normalizedPath) {
if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue;
}
const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path;
const nextSegment = relativePath.split("/")[0];
if (!nextSegment) continue;
const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment;
if (children.has(childPath)) continue;
const childEntry = this.#entries.get(childPath);
const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/");
children.set(childPath, {
name: nextSegment,
path: childPath,
isDirectory,
size: isDirectory ? 0 : (childEntry?.size ?? entry.size),
mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs,
});
}
return [...children.values()].sort((left, right) =>
left.name.toLowerCase().localeCompare(right.name.toLowerCase()),
);
}
async readFile(subPath: string): Promise<ExtractedArchiveFile> {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (!normalizedPath) {
throw new ToolError("Archive file path is required");
}
const entry = this.#entries.get(normalizedPath);
if (!entry) {
throw new ToolError(`Archive file '${normalizedPath}' not found`);
}
if (entry.isDirectory) {
throw new ToolError(`Archive path '${normalizedPath}' is a directory`);
}
if (!entry.storage) {
throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`);
}
if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) {
throw new ToolError(
`Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`,
);
}
const bytes =
entry.storage.type === "tar"
? await entry.storage.file.bytes()
: await readZipFileBytes(entry.storage, entry.size);
return {
path: entry.path,
isDirectory: false,
size: entry.size,
mtimeMs: entry.mtimeMs,
bytes,
};
}
}
export async function openArchive(filePath: string): Promise<ArchiveReader> {
const format = getArchiveFormatFromPath(filePath);
if (!format) {
throw new ToolError(`Unsupported archive format: ${filePath}`);
}
if (format === "zip") {
return new ArchiveReader(format, await readZipEntries(filePath));
}
const file = Bun.file(filePath);
const archiveSize = file.size;
if (archiveSize > MAX_TAR_ARCHIVE_BYTES) {
throw new ToolError(
`Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
);
}
const entries = await readTarEntries(await file.bytes());
return new ArchiveReader(format, entries);
}
export async function listArchiveRoot(
bytes: Uint8Array,
format: ArchiveFormat,
opts: { limit?: number } = {},
): Promise<string> {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-archive-"));
const tempPath = path.join(tempDir, `payload.${format}`);
try {
await Bun.write(tempPath, bytes);
const archive = await openArchive(tempPath);
const entries = archive.listDirectory("");
const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries;
const lines = formatArchiveEntryLines(limitedEntries);
return lines.length > 0 ? lines.join("\n") : "(empty archive directory)";
} finally {
await fs.rm(tempDir, { recursive: true, force: true });
}
}
+1 -1
View File
@@ -20,12 +20,12 @@ import { CachedOutputBlock, markFramedBlockComponent } from "../tui/output-block
import { webpExclusionForModel } from "../utils/image-loading";
import { formatDimensionNote, resizeImage } from "../utils/image-resize";
import { ensureTool } from "../utils/tools-manager";
import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "../utils/zip";
import { extractWithParallel, findParallelApiKey, getParallelExtractContent } from "../web/parallel";
import { specialHandlers } from "../web/scrapers";
import type { RenderResult } from "../web/scrapers/types";
import { finalizeOutput, loadPage, looksLikeHtml, MAX_BYTES, MAX_OUTPUT_CHARS } from "../web/scrapers/types";
import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils";
import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "./archive-reader";
import { applyListLimit } from "./list-limit";
import { formatStyledArtifactReference, type OutputMeta } from "./output-meta";
import { type LineRange, parseLineRanges } from "./path-utils";
+1 -1
View File
@@ -48,8 +48,8 @@ import {
webpExclusionForModel,
} from "../utils/image-loading";
import { convertFileWithMarkit } from "../utils/markit";
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "../utils/zip";
import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree";
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader";
import {
type ConflictEntry,
type ConflictScope,
+1 -6
View File
@@ -28,13 +28,8 @@ import {
uriHyperlink,
} from "../tui";
import { resolveFileDisplayMode } from "../utils/file-display-mode";
import { type ArchiveReader, type ExtractedArchiveFile, openArchive, parseArchivePathCandidates } from "../utils/zip";
import type { ToolSession } from ".";
import {
type ArchiveReader,
type ExtractedArchiveFile,
openArchive,
parseArchivePathCandidates,
} from "./archive-reader";
import { createFileRecorder, formatResultPath } from "./file-recorder";
import { classifyGroupedLines, formatGroupedFiles, groupLineIndicesByBlank } from "./grouped-file-output";
import { formatMatchLine } from "./match-line-format";
+25 -59
View File
@@ -20,8 +20,14 @@ import writeDescription from "../prompts/tools/write.md" with { type: "text" };
import type { ToolSession } from "../sdk";
import { fileHyperlink, framedBlock, renderStatusLine } from "../tui";
import { resolveFileDisplayMode } from "../utils/file-display-mode";
import {
type ArchiveMemberContent,
archiveFormatFromPath,
parseArchivePathCandidates,
readArchiveEntries,
writeArchive,
} from "../utils/zip";
import { truncateForPrompt } from "./approval";
import { parseArchivePathCandidates } from "./archive-reader";
import { assertEditableFile } from "./auto-generated-guard";
import {
type ConflictEntry,
@@ -363,9 +369,10 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
const finalPath = resolvedArchivePath.exists
? await fs.realpath(resolvedArchivePath.absolutePath).catch(() => resolvedArchivePath.absolutePath)
: resolvedArchivePath.absolutePath;
const lowerPath = finalPath.toLowerCase();
const isZip = lowerPath.endsWith(".zip");
const isGzip = lowerPath.endsWith(".tar.gz") || lowerPath.endsWith(".tgz");
// A realpath swap can land on a name without an archive extension; a
// whole-archive rewrite then defaults to an uncompressed tar, matching the
// previous `isZip`/`isGzip`/else fallthrough.
const format = archiveFormatFromPath(finalPath) ?? "tar";
// Rewrites are whole-archive: write to a temp file and rename so a
// crash/disk-full mid-write can't destroy the original archive.
const tmpPath = `${finalPath}.tmp-${process.pid}`;
@@ -375,66 +382,25 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
await fs.mkdir(parentDir, { recursive: true });
}
if (isZip) {
const zipEntries: Record<string, Uint8Array> = {};
if (resolvedArchivePath.exists) {
try {
const bytes = await Bun.file(resolvedArchivePath.absolutePath).bytes();
const { unzip } = await import("../utils/zip");
const existing = unzip(new Uint8Array(bytes));
for (const [entryPath, data] of Object.entries(existing)) {
zipEntries[entryPath.replace(/\\/g, "/")] = data;
}
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
}
zipEntries[resolvedArchivePath.archiveSubPath] = new TextEncoder().encode(content);
const entries = new Map<string, ArchiveMemberContent>();
if (resolvedArchivePath.exists) {
try {
const { zip } = await import("../utils/zip");
const zipBuffer = zip(zipEntries);
await Bun.write(tmpPath, zipBuffer);
await fs.rename(tmpPath, finalPath);
const existing = await readArchiveEntries({ bytes: await Bun.file(finalPath).bytes(), format });
for (const [entryPath, data] of existing) {
entries.set(entryPath, data);
}
} catch (error) {
await fs.rm(tmpPath, { force: true }).catch(() => {});
throw new ToolError(error instanceof Error ? error.message : String(error));
}
} else {
const archiveEntries: Record<string, string | File> = {};
if (resolvedArchivePath.exists) {
let archive: Bun.Archive;
try {
archive = new Bun.Archive(await Bun.file(resolvedArchivePath.absolutePath).bytes());
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
}
entries.set(resolvedArchivePath.archiveSubPath, content);
let files: Map<string, File>;
try {
files = await archive.files();
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
for (const [entryPath, file] of files) {
archiveEntries[entryPath.replace(/\\/g, "/")] = file;
}
}
archiveEntries[resolvedArchivePath.archiveSubPath] = content;
try {
// `Bun.Archive.write` never infers compression from the extension;
// request gzip explicitly so `.tar.gz`/`.tgz` stay compressed.
await Bun.Archive.write(tmpPath, archiveEntries, isGzip ? { compress: "gzip" } : undefined);
await fs.rename(tmpPath, finalPath);
} catch (error) {
await fs.rm(tmpPath, { force: true }).catch(() => {});
throw new ToolError(error instanceof Error ? error.message : String(error));
}
try {
await writeArchive(tmpPath, format, entries);
await fs.rename(tmpPath, finalPath);
} catch (error) {
await fs.rm(tmpPath, { force: true }).catch(() => {});
throw new ToolError(error instanceof Error ? error.message : String(error));
}
invalidateFsScanAfterWrite(resolvedArchivePath.absolutePath);
@@ -2,6 +2,7 @@ import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { $which, APP_NAME, getToolsDir, logger, ptree, TempDir } from "@oh-my-pi/pi-utils";
import { extractArchive } from "./zip";
const TOOLS_DIR = getToolsDir();
const TOOL_DOWNLOAD_TIMEOUT_MS = 120_000;
@@ -220,17 +221,7 @@ async function downloadTool(tool: ToolName, signal?: AbortSignal): Promise<strin
}
try {
const archive = new Bun.Archive(await Bun.file(archivePath).arrayBuffer());
const files = await archive.files();
const extractRoot = path.resolve(tmp.path());
for (const [filePath, file] of files) {
const outputPath = path.resolve(extractRoot, filePath);
if (!outputPath.startsWith(extractRoot + path.sep)) {
throw new Error(`Archive entry escapes extraction dir: ${filePath}`);
}
await Bun.write(outputPath, file);
}
await extractArchive(archivePath, tmp.path());
} catch (err) {
throw new Error(`Failed to extract ${assetName}: ${err instanceof Error ? err.message : String(err)}`);
}
+873 -11
View File
@@ -1,29 +1,891 @@
// The single ZIP/DEFLATE boundary for the codebase. This is the ONLY module
// that imports `fflate`; the markit document converters, the write tool, and
// the archive reader all go through here so there is exactly one ZIP
// implementation to reason about. Do not import `fflate` (or another archive
// library) anywhere else.
// The single archive boundary for the codebase: ZIP (via `fflate`) and tar /
// tar.gz (via `Bun.Archive`), built on the raw DEFLATE helpers the ZIP reader
// needs. This is the ONLY module that imports `fflate` or touches
// `Bun.Archive`; the markit document converters, the read/search/write tools,
// the URL fetcher, the debug report bundler, and the tool-binary installer all
// go through here so there is exactly one archive implementation to reason
// about. Do not import `fflate` (or call `Bun.Archive`) anywhere else.
import * as path from "node:path";
import { formatBytes } from "@oh-my-pi/pi-utils";
import type { Unzipped } from "fflate";
import { inflateSync, strFromU8 } from "fflate";
import { inflateSync, unzipSync, zipSync } from "fflate";
import { ToolError } from "../tools/tool-errors";
export type { Unzipped } from "fflate";
export { unzipSync as unzip, zipSync as zip } from "fflate";
const ENCODER = new TextEncoder();
// fflate's `strFromU8` is just `new TextDecoder().decode()` for UTF-8, so the
// converters and ZIP filename reader use the platform decoders directly.
const UTF8_DECODER = new TextDecoder();
// ZIP central-directory names without the UTF-8 flag carry no reliable encoding;
// decode them as their legacy code page (windows-1252) as a stable best effort.
const LEGACY_NAME_DECODER = new TextDecoder("windows-1252");
/** Read a single ZIP entry as UTF-8 text, or `undefined` when the entry is absent. */
export function unzipText(entries: Unzipped, entryPath: string): string | undefined {
const data = entries[entryPath];
return data ? strFromU8(data) : undefined;
return data ? UTF8_DECODER.decode(data) : undefined;
}
/**
* Inflate a raw DEFLATE stream (a single deflate-compressed ZIP member). Pass a
* preallocated `into` buffer when the uncompressed size is known up front.
*/
export function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
return into ? inflateSync(bytes, { out: into }) : inflateSync(bytes);
}
/** Decode raw bytes as text — UTF-8 by default, latin1 when `latin1` is set. */
export function bytesToText(bytes: Uint8Array, latin1?: boolean): string {
return strFromU8(bytes, latin1);
/**
* Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into
* memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is
* exempt: it is read via ranged central-directory access.
*/
const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024;
/**
* Cap on a single archive member's declared (uncompressed) size. The declared
* size is attacker-controlled metadata — a crafted ZIP entry can claim
* multi-GB sizes that would be allocated up front before any data inflates.
*/
const MAX_ARCHIVE_MEMBER_BYTES = 64 * 1024 * 1024;
export type ArchiveFormat = "zip" | "tar" | "tar.gz";
/**
* Where to read an archive from: a filesystem path (format inferred from the
* extension; ZIP is read lazily via ranged central-directory access) or
* in-memory bytes with an explicit format.
*/
export type ArchiveSource = string | { bytes: Uint8Array; format: ArchiveFormat };
/** Content for a member when packing or extracting an archive. */
export type ArchiveMemberContent = string | Uint8Array | Blob;
export interface ArchivePathCandidate {
archivePath: string;
subPath: string;
}
export interface ArchiveNode {
path: string;
isDirectory: boolean;
size: number;
mtimeMs?: number;
}
export interface ArchiveDirectoryEntry extends ArchiveNode {
name: string;
}
export interface ExtractedArchiveFile extends ArchiveNode {
bytes: Uint8Array;
}
/** A byte window into an archive — file-backed (lazy) or in-memory. */
interface ByteSource {
readonly size: number;
read(start: number, end: number): Promise<Uint8Array>;
}
function assertValidRange(start: number, end: number): void {
if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) {
throw new ToolError("Invalid ZIP archive range");
}
}
function fileByteSource(filePath: string): ByteSource {
const file = Bun.file(filePath);
const size = file.size;
if (!Number.isSafeInteger(size)) {
throw new ToolError("ZIP archive is too large to read safely");
}
return {
size,
async read(start, end) {
assertValidRange(start, end);
const bytes = await file.slice(start, end).bytes();
if (bytes.byteLength !== end - start) {
throw new ToolError("Invalid ZIP archive: truncated data");
}
return bytes;
},
};
}
function memoryByteSource(buffer: Uint8Array): ByteSource {
return {
size: buffer.byteLength,
async read(start, end) {
assertValidRange(start, end);
if (end > buffer.byteLength) {
throw new ToolError("Invalid ZIP archive: truncated data");
}
return buffer.subarray(start, end);
},
};
}
interface TarStorage {
type: "tar";
file: File;
}
interface ZipStorage {
type: "zip";
source: ByteSource;
compressedSize: number;
compression: number;
flags: number;
localHeaderOffset: number;
}
type EntryStorage = TarStorage | ZipStorage;
interface ArchiveIndexEntry extends ArchiveNode {
storage?: EntryStorage;
}
function normalizeArchiveLookupPath(rawPath?: string): string | undefined {
if (!rawPath) return "";
const parts = rawPath.replace(/\\/g, "/").split("/");
const normalizedParts: string[] = [];
for (const part of parts) {
if (!part || part === ".") continue;
if (part === "..") return undefined;
normalizedParts.push(part);
}
return normalizedParts.join("/");
}
function normalizeArchiveEntryPath(rawPath: string): string | undefined {
const parts = rawPath.replace(/\\/g, "/").split("/");
const normalizedParts: string[] = [];
for (const part of parts) {
if (!part || part === ".") continue;
if (part === "..") return undefined;
normalizedParts.push(part);
}
if (normalizedParts.length === 0) return undefined;
return normalizedParts.join("/");
}
function isArchiveDirectoryName(rawPath: string): boolean {
return rawPath.endsWith("/") || rawPath.endsWith("\\");
}
function upsertArchiveEntry(map: Map<string, ArchiveIndexEntry>, entry: ArchiveIndexEntry): void {
const existing = map.get(entry.path);
if (!existing) {
map.set(entry.path, entry);
return;
}
if (existing.isDirectory && !entry.isDirectory) {
map.set(entry.path, entry);
return;
}
if (!existing.isDirectory && entry.isDirectory) {
return;
}
map.set(entry.path, {
...existing,
size: existing.size || entry.size,
mtimeMs: existing.mtimeMs ?? entry.mtimeMs,
storage: existing.storage ?? entry.storage,
});
}
function ensureParentDirectories(map: Map<string, ArchiveIndexEntry>): void {
for (const entry of [...map.values()]) {
const parts = entry.path.split("/");
const stop = parts.length - 1;
for (let index = 1; index <= stop; index++) {
const dirPath = parts.slice(0, index).join("/");
if (!dirPath || map.has(dirPath)) continue;
map.set(dirPath, {
path: dirPath,
isDirectory: true,
size: 0,
});
}
}
}
/** Infer an archive format from a filesystem path's extension. */
export function archiveFormatFromPath(filePath: string): ArchiveFormat | undefined {
const normalized = filePath.toLowerCase();
if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz";
if (normalized.endsWith(".tar")) return "tar";
if (normalized.endsWith(".zip")) return "zip";
return undefined;
}
export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] {
return entries.map(entry => {
if (entry.isDirectory) return `${entry.name}/`;
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
return `${entry.name}${sizeSuffix}`;
});
}
const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50;
const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50;
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50;
const ZIP_EOCD_SIGNATURE = 0x06054b50;
const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50;
const ZIP_EOCD_MIN_LENGTH = 22;
const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff;
const ZIP64_EOCD_LOCATOR_LENGTH = 20;
const ZIP_STORED_COMPRESSION = 0;
const ZIP_DEFLATE_COMPRESSION = 8;
const ZIP_UTF8_FLAG = 0x0800;
const ZIP_ENCRYPTED_FLAG = 0x0001;
const ZIP_UINT16_MAX = 0xffff;
const ZIP_UINT32_MAX = 0xffffffff;
const ZIP_UINT32_RANGE = 0x100000000;
interface ZipCentralDirectoryInfo {
entries: number;
offset: number;
size: number;
}
interface Zip64EntryValues {
compressedSize: number;
uncompressedSize: number;
localHeaderOffset: number;
diskStart: number;
}
interface Zip64EntryPlaceholders {
compressedSize: boolean;
uncompressedSize: boolean;
localHeaderOffset: boolean;
diskStart: boolean;
}
function readUInt16LE(bytes: Uint8Array, offset: number): number {
return bytes[offset]! | (bytes[offset + 1]! << 8);
}
function readUInt32LE(bytes: Uint8Array, offset: number): number {
return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0;
}
function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean {
if (bytes.byteLength < offset + value.length) return false;
for (let index = 0; index < value.length; index++) {
if (bytes[offset + index] !== value.charCodeAt(index)) return false;
}
return true;
}
export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined {
if (bytes.byteLength >= 4) {
const signature = readUInt32LE(bytes, 0);
if (
signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE ||
signature === ZIP_EOCD_SIGNATURE ||
signature === ZIP_DATA_DESCRIPTOR_SIGNATURE
) {
return "zip";
}
}
if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) {
return "tar.gz";
}
if (bytesMatchAscii(bytes, 257, "ustar")) {
return "tar";
}
return undefined;
}
function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number {
const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE;
if (!Number.isSafeInteger(value)) {
throw new ToolError("ZIP archive uses offsets or sizes too large to read safely");
}
return value;
}
function findEndOfCentralDirectory(tail: Uint8Array): number {
for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) {
if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue;
const commentLength = readUInt16LE(tail, offset + 20);
if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset;
}
throw new ToolError("Invalid ZIP archive: missing end of central directory");
}
async function readZip64CentralDirectoryInfo(
source: ByteSource,
tail: Uint8Array,
tailStart: number,
eocdOffset: number,
): Promise<ZipCentralDirectoryInfo | undefined> {
const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH;
if (locatorOffset < 0) return undefined;
const locator =
locatorOffset >= tailStart
? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH)
: await source.read(locatorOffset, eocdOffset);
if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined;
const zip64EocdDisk = readUInt32LE(locator, 4);
const zip64EocdOffset = readUInt64LEAsNumber(locator, 8);
const totalDisks = readUInt32LE(locator, 16);
if (zip64EocdDisk !== 0 || totalDisks > 1) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
const record = await source.read(zip64EocdOffset, zip64EocdOffset + 56);
if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory");
}
if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
return {
entries: readUInt64LEAsNumber(record, 32),
size: readUInt64LEAsNumber(record, 40),
offset: readUInt64LEAsNumber(record, 48),
};
}
async function readZipCentralDirectoryInfo(source: ByteSource): Promise<ZipCentralDirectoryInfo> {
const fileSize = source.size;
if (fileSize < ZIP_EOCD_MIN_LENGTH) {
throw new ToolError("Invalid ZIP archive: missing end of central directory");
}
const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH);
const tailStart = fileSize - tailLength;
const tail = await source.read(tailStart, fileSize);
const eocdIndex = findEndOfCentralDirectory(tail);
const eocdOffset = tailStart + eocdIndex;
if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
let entries = readUInt16LE(tail, eocdIndex + 10);
let size = readUInt32LE(tail, eocdIndex + 12);
let offset = readUInt32LE(tail, eocdIndex + 16);
const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX;
const zip64Info = await readZip64CentralDirectoryInfo(source, tail, tailStart, eocdOffset);
if (zip64Info) {
({ entries, size, offset } = zip64Info);
} else if (needsZip64) {
throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata");
}
if (offset + size > fileSize) {
throw new ToolError("Invalid ZIP archive: central directory exceeds file size");
}
return { entries, offset, size };
}
function readZip64EntryValues(
extra: Uint8Array,
placeholders: Zip64EntryPlaceholders,
current: Zip64EntryValues,
): Zip64EntryValues {
if (
!placeholders.compressedSize &&
!placeholders.uncompressedSize &&
!placeholders.localHeaderOffset &&
!placeholders.diskStart
) {
return current;
}
let offset = 0;
while (offset + 4 <= extra.byteLength) {
const headerId = readUInt16LE(extra, offset);
const dataSize = readUInt16LE(extra, offset + 2);
const dataStart = offset + 4;
const dataEnd = dataStart + dataSize;
if (dataEnd > extra.byteLength) {
throw new ToolError("Invalid ZIP archive: malformed extra field");
}
if (headerId === 0x0001) {
let cursor = dataStart;
let uncompressedSize = current.uncompressedSize;
let compressedSize = current.compressedSize;
let localHeaderOffset = current.localHeaderOffset;
let diskStart = current.diskStart;
if (placeholders.uncompressedSize) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
uncompressedSize = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.compressedSize) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
compressedSize = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.localHeaderOffset) {
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
localHeaderOffset = readUInt64LEAsNumber(extra, cursor);
cursor += 8;
}
if (placeholders.diskStart) {
if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
diskStart = readUInt32LE(extra, cursor);
}
return { compressedSize, uncompressedSize, localHeaderOffset, diskStart };
}
offset = dataEnd;
}
throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field");
}
function parseZipCentralDirectory(
source: ByteSource,
centralDirectory: Uint8Array,
expectedEntries: number,
): ArchiveIndexEntry[] {
const entries: ArchiveIndexEntry[] = [];
let offset = 0;
for (let index = 0; index < expectedEntries; index++) {
if (offset + 46 > centralDirectory.byteLength) {
throw new ToolError("Invalid ZIP archive: truncated central directory");
}
if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: malformed central directory");
}
const flags = readUInt16LE(centralDirectory, offset + 8);
const compression = readUInt16LE(centralDirectory, offset + 10);
const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20);
const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24);
const fileNameLength = readUInt16LE(centralDirectory, offset + 28);
const extraLength = readUInt16LE(centralDirectory, offset + 30);
const commentLength = readUInt16LE(centralDirectory, offset + 32);
const diskStartRaw = readUInt16LE(centralDirectory, offset + 34);
const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42);
const nameStart = offset + 46;
const extraStart = nameStart + fileNameLength;
const entryEnd = extraStart + extraLength + commentLength;
if (entryEnd > centralDirectory.byteLength) {
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
}
const useLegacyEncoding = (flags & ZIP_UTF8_FLAG) === 0;
const rawPath = (useLegacyEncoding ? LEGACY_NAME_DECODER : UTF8_DECODER).decode(
centralDirectory.subarray(nameStart, extraStart),
);
const normalizedPath = normalizeArchiveEntryPath(rawPath);
if (normalizedPath) {
const values = readZip64EntryValues(
centralDirectory.subarray(extraStart, extraStart + extraLength),
{
compressedSize: compressedSizeRaw === ZIP_UINT32_MAX,
uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX,
localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX,
diskStart: diskStartRaw === ZIP_UINT16_MAX,
},
{
compressedSize: compressedSizeRaw,
uncompressedSize: uncompressedSizeRaw,
localHeaderOffset: localHeaderOffsetRaw,
diskStart: diskStartRaw,
},
);
if (values.diskStart !== 0) {
throw new ToolError("Multi-disk ZIP archives are not supported");
}
const isDirectory = isArchiveDirectoryName(rawPath);
entries.push({
path: normalizedPath,
isDirectory,
size: isDirectory ? 0 : values.uncompressedSize,
storage: isDirectory
? undefined
: {
type: "zip",
source,
compressedSize: values.compressedSize,
compression,
flags,
localHeaderOffset: values.localHeaderOffset,
},
});
}
offset = entryEnd;
}
return entries;
}
async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise<Uint8Array> {
if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) {
throw new ToolError("Encrypted ZIP entries are not supported");
}
const localHeader = await storage.source.read(storage.localHeaderOffset, storage.localHeaderOffset + 30);
if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) {
throw new ToolError("Invalid ZIP archive: malformed local file header");
}
const fileNameLength = readUInt16LE(localHeader, 26);
const extraLength = readUInt16LE(localHeader, 28);
const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength;
const compressedBytes = await storage.source.read(dataStart, dataStart + storage.compressedSize);
if (storage.compression === ZIP_STORED_COMPRESSION) {
return compressedBytes;
}
if (storage.compression !== ZIP_DEFLATE_COMPRESSION) {
throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`);
}
try {
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
}
async function readTarEntries(bytes: Uint8Array): Promise<ArchiveIndexEntry[]> {
let archive: Bun.Archive;
try {
archive = new Bun.Archive(bytes);
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
let files: Map<string, File>;
try {
files = await archive.files();
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
const entries: ArchiveIndexEntry[] = [];
for (const [rawPath, file] of files) {
const normalizedPath = normalizeArchiveEntryPath(rawPath);
if (!normalizedPath) continue;
const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined;
entries.push({
path: normalizedPath,
isDirectory: false,
size: file.size,
mtimeMs,
storage: { type: "tar", file },
});
}
return entries;
}
async function readZipEntries(source: ByteSource): Promise<ArchiveIndexEntry[]> {
const directoryInfo = await readZipCentralDirectoryInfo(source);
const centralDirectory = await source.read(directoryInfo.offset, directoryInfo.offset + directoryInfo.size);
return parseZipCentralDirectory(source, centralDirectory, directoryInfo.entries);
}
/**
* Split an `archive.ext:inner/path` reference into every plausible
* `{ archivePath, subPath }` pair, longest archive prefix first. A path may
* contain more than one archive extension, so each candidate is a guess at
* where the archive ends and the member portion begins.
*/
export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] {
const normalized = filePath.replace(/\\/g, "/");
const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi;
const seen = new Set<string>();
const candidates: ArchivePathCandidate[] = [];
let match: RegExpExecArray | null;
while (true) {
match = pattern.exec(normalized);
if (match === null) {
break;
}
const end = match.index + match[0].length;
const archivePath = filePath.slice(0, end);
const subPath = normalized.slice(end).replace(/^:+/, "");
const key = `${archivePath}\0${subPath}`;
if (seen.has(key)) continue;
seen.add(key);
candidates.push({ archivePath, subPath });
}
return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length);
}
/**
* An indexed, read-only view over a single archive. ZIP archives are indexed
* from the central directory and members are inflated on demand; tar archives
* are fully materialized by `Bun.Archive` up front.
*/
export class ArchiveReader {
readonly format: ArchiveFormat;
#entries = new Map<string, ArchiveIndexEntry>();
constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) {
this.format = format;
for (const entry of entries) {
upsertArchiveEntry(this.#entries, entry);
}
ensureParentDirectories(this.#entries);
}
getNode(subPath?: string): ArchiveNode | undefined {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (normalizedPath === undefined) return undefined;
if (normalizedPath === "") {
return { path: "", isDirectory: true, size: 0 };
}
const entry = this.#entries.get(normalizedPath);
if (!entry) return undefined;
return {
path: entry.path,
isDirectory: entry.isDirectory,
size: entry.size,
mtimeMs: entry.mtimeMs,
};
}
listDirectory(subPath?: string): ArchiveDirectoryEntry[] {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (normalizedPath === undefined) {
throw new ToolError("Archive path cannot contain '..'");
}
if (normalizedPath) {
const entry = this.#entries.get(normalizedPath);
if (!entry) {
throw new ToolError(`Archive path '${normalizedPath}' not found`);
}
if (!entry.isDirectory) {
throw new ToolError(`Archive path '${normalizedPath}' is not a directory`);
}
}
const prefix = normalizedPath ? `${normalizedPath}/` : "";
const children = new Map<string, ArchiveDirectoryEntry>();
for (const entry of this.#entries.values()) {
if (normalizedPath) {
if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue;
}
const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path;
const nextSegment = relativePath.split("/")[0];
if (!nextSegment) continue;
const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment;
if (children.has(childPath)) continue;
const childEntry = this.#entries.get(childPath);
const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/");
children.set(childPath, {
name: nextSegment,
path: childPath,
isDirectory,
size: isDirectory ? 0 : (childEntry?.size ?? entry.size),
mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs,
});
}
return [...children.values()].sort((left, right) =>
left.name.toLowerCase().localeCompare(right.name.toLowerCase()),
);
}
async readFile(subPath: string): Promise<ExtractedArchiveFile> {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (!normalizedPath) {
throw new ToolError("Archive file path is required");
}
const entry = this.#entries.get(normalizedPath);
if (!entry) {
throw new ToolError(`Archive file '${normalizedPath}' not found`);
}
if (entry.isDirectory) {
throw new ToolError(`Archive path '${normalizedPath}' is a directory`);
}
if (!entry.storage) {
throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`);
}
if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) {
throw new ToolError(
`Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`,
);
}
const bytes =
entry.storage.type === "tar"
? await entry.storage.file.bytes()
: await readZipFileBytes(entry.storage, entry.size);
return {
path: entry.path,
isDirectory: false,
size: entry.size,
mtimeMs: entry.mtimeMs,
bytes,
};
}
}
/**
* Open an archive for reading. ZIP archives opened from a path are indexed
* lazily via ranged central-directory reads (members inflate on demand); tar
* archives and in-memory ZIPs are read from a single buffer.
*/
export async function openArchive(source: ArchiveSource): Promise<ArchiveReader> {
if (typeof source === "string") {
const format = archiveFormatFromPath(source);
if (!format) {
throw new ToolError(`Unsupported archive format: ${source}`);
}
if (format === "zip") {
return new ArchiveReader(format, await readZipEntries(fileByteSource(source)));
}
const file = Bun.file(source);
const archiveSize = file.size;
if (archiveSize > MAX_TAR_ARCHIVE_BYTES) {
throw new ToolError(
`Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
);
}
return new ArchiveReader(format, await readTarEntries(await file.bytes()));
}
const { bytes, format } = source;
if (format === "zip") {
return new ArchiveReader(format, await readZipEntries(memoryByteSource(bytes)));
}
if (bytes.byteLength > MAX_TAR_ARCHIVE_BYTES) {
throw new ToolError(
`Archive is too large to read in memory (${formatBytes(bytes.byteLength)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
);
}
return new ArchiveReader(format, await readTarEntries(bytes));
}
/** Render the top-level entries of an in-memory archive as one line each. */
export async function listArchiveRoot(
bytes: Uint8Array,
format: ArchiveFormat,
opts: { limit?: number } = {},
): Promise<string> {
const archive = await openArchive({ bytes, format });
const entries = archive.listDirectory("");
const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries;
const lines = formatArchiveEntryLines(limitedEntries);
return lines.length > 0 ? lines.join("\n") : "(empty archive directory)";
}
async function resolveArchiveBytes(source: ArchiveSource): Promise<{ bytes: Uint8Array; format: ArchiveFormat }> {
if (typeof source !== "string") return source;
const format = archiveFormatFromPath(source);
if (!format) {
throw new ToolError(`Unsupported archive format: ${source}`);
}
return { bytes: await Bun.file(source).bytes(), format };
}
async function memberToBytes(content: ArchiveMemberContent): Promise<Uint8Array> {
if (typeof content === "string") return ENCODER.encode(content);
if (content instanceof Uint8Array) return content;
return new Uint8Array(await content.arrayBuffer());
}
/**
* Fully materialize every file member into a `path → content` map: ZIP members
* are inflated via fflate, tar members are returned as lazy `File`s. Use this
* when you need every entry (rewrite, extract); for browsing or single-member
* reads prefer `openArchive`, which is lazy for ZIP.
*/
export async function readArchiveEntries(source: ArchiveSource): Promise<Map<string, ArchiveMemberContent>> {
const { bytes, format } = await resolveArchiveBytes(source);
const entries = new Map<string, ArchiveMemberContent>();
if (format === "zip") {
const unzipped = unzipSync(bytes);
for (const name in unzipped) {
entries.set(name.replace(/\\/g, "/"), unzipped[name]!);
}
return entries;
}
const files = await new Bun.Archive(bytes).files();
for (const [name, file] of files) {
entries.set(name.replace(/\\/g, "/"), file);
}
return entries;
}
/**
* Serialize `entries` into an archive of `format` and write it to `destPath`.
* ZIP is built with fflate, tar / tar.gz with `Bun.Archive` (gzip for tar.gz).
* String members are encoded as UTF-8.
*/
export async function writeArchive(
destPath: string,
format: ArchiveFormat,
entries: Iterable<readonly [string, ArchiveMemberContent]>,
): Promise<void> {
if (format === "zip") {
const record: Record<string, Uint8Array> = {};
for (const [name, content] of entries) {
record[name.replace(/\\/g, "/")] = await memberToBytes(content);
}
await Bun.write(destPath, zipSync(record));
return;
}
const record: Record<string, ArchiveMemberContent> = {};
for (const [name, content] of entries) {
record[name.replace(/\\/g, "/")] = content;
}
await Bun.Archive.write(destPath, record, format === "tar.gz" ? { compress: "gzip" } : undefined);
}
/**
* Extract every file member to `destDir`, creating parent directories as
* needed. Entries that would escape `destDir` (via `..` or an absolute path)
* are rejected. Returns the number of files written.
*/
export async function extractArchive(source: ArchiveSource, destDir: string): Promise<number> {
const extractRoot = path.resolve(destDir);
const entries = await readArchiveEntries(source);
let count = 0;
for (const [name, content] of entries) {
if (name.endsWith("/")) continue;
const outputPath = path.resolve(extractRoot, name);
if (!outputPath.startsWith(extractRoot + path.sep)) {
throw new ToolError(`Archive entry escapes extraction dir: ${name}`);
}
await Bun.write(outputPath, content);
count++;
}
return count;
}