feat(coding-agent): consolidated and optimize archive handling
- Centralized archive operations into a new `utils/zip.ts` module with unified support for ZIP, tar, and tar.gz formats. - Optimized ZIP reading using lazy, ranged central-directory access and implemented ZIP64 support for large files. - Hardened archive extraction with directory traversal protection and configured memory limits for loading and extraction. - Refactored tool-specific logic to utilize the new centralized utility and deleted the redundant `archive-reader.ts`.
This commit is contained in:
+1
-1
@@ -7,7 +7,7 @@
|
||||
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/read.md`
|
||||
- Key collaborators:
|
||||
- `packages/coding-agent/src/tools/path-utils.ts` — split `path` from trailing selectors; normalize local paths.
|
||||
- `packages/coding-agent/src/tools/archive-reader.ts` — detect `archive.ext:inner/path`, index archives, list/read entries.
|
||||
- `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: detect `archive.ext:inner/path`, index archives, list/read entries.
|
||||
- `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite targets, parse selectors, render tables.
|
||||
- `packages/coding-agent/src/tools/fetch.ts` — URL parsing, fetch/render pipeline, URL cache/artifacts.
|
||||
- `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`.
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
- Entry: `packages/coding-agent/src/tools/write.ts`
|
||||
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/write.md`
|
||||
- Key collaborators:
|
||||
- `packages/coding-agent/src/tools/archive-reader.ts` — parse `archive.ext:entry` selectors.
|
||||
- `packages/coding-agent/src/utils/zip.ts` — the unified ZIP/tar wrapper: parse `archive.ext:entry` selectors and rewrite the archive whole.
|
||||
- `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite paths and perform row insert/update/delete.
|
||||
- `packages/coding-agent/src/tools/conflict-detect.ts` — parse `conflict://` URIs and splice recorded merge-conflict regions.
|
||||
- `packages/coding-agent/src/lsp/index.ts` — format-on-write and diagnostics writethrough.
|
||||
@@ -56,7 +56,7 @@ Single-shot result.
|
||||
1. `WriteTool.execute()` in `packages/coding-agent/src/tools/write.ts` strips pasted `[PATH#HASH]` headers and `LINE:` hashline prefixes from `content` when the session is in hashline display mode.
|
||||
2. If `path` is an internal URL whose handler exposes `write`, the tool delegates directly to `handler.write(...)` and returns.
|
||||
3. `conflict://...` paths are handled next by the merge-conflict resolver. Scope reads such as `conflict://<id>/ours` are rejected as read-only; writable conflict URIs must omit the scope.
|
||||
4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/tools/archive-reader.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet.
|
||||
4. It calls `#resolveArchiveWritePath()` next. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/utils/zip.ts`, checks candidate archive files on disk (longest match first), and falls back to the shortest candidate archive path even when the archive file does not exist yet.
|
||||
5. Archive writes call `enforcePlanModeWrite(..., { op: exists ? "update" : "create" })`, then `#writeArchiveEntry()`.
|
||||
- The parent directory of the archive file is created with `fs.mkdir(..., { recursive: true })`.
|
||||
- `.zip` archives are read with `fflate.unzipSync()`, the target entry is replaced in an in-memory map, and the archive is rewritten with `fflate.zipSync()` + `Bun.write()`.
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Refactored internal archive handling into a unified `src/utils/zip.ts` module
|
||||
- Centralized all `fflate` (ZIP) and `Bun.Archive` (tar/tar.gz) operations into `zip.ts`
|
||||
- Optimized archive reading by using lazy, ranged central-directory access for ZIP files
|
||||
- Updated internal image processing to no longer include metadata text for fetched images
|
||||
- Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob
|
||||
- Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved.
|
||||
|
||||
@@ -7,6 +7,7 @@ import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import type { WorkProfile } from "@oh-my-pi/pi-natives";
|
||||
import { APP_NAME, getLogPath, getLogsDir, getReportsDir, isEnoent } from "@oh-my-pi/pi-utils";
|
||||
import { writeArchive } from "../utils/zip";
|
||||
import type { CpuProfile, HeapSnapshot } from "./profiler";
|
||||
import { collectSystemInfo, sanitizeEnv } from "./system-info";
|
||||
|
||||
@@ -165,7 +166,7 @@ export async function createReportBundle(options: ReportBundleOptions): Promise<
|
||||
}
|
||||
|
||||
// Write archive
|
||||
await Bun.Archive.write(outputPath, data, { compress: "gzip" });
|
||||
await writeArchive(outputPath, "tar.gz", Object.entries(data));
|
||||
|
||||
return { path: outputPath, files };
|
||||
}
|
||||
|
||||
@@ -1,721 +0,0 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { bytesToText, inflateRaw } from "../utils/zip";
|
||||
|
||||
import { formatBytes } from "./render-utils";
|
||||
import { ToolError } from "./tool-errors";
|
||||
|
||||
/**
|
||||
* Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into
|
||||
* memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is
|
||||
* exempt: it is read via ranged central-directory access.
|
||||
*/
|
||||
const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024;
|
||||
/**
|
||||
* Cap on a single archive member's declared (uncompressed) size. The declared
|
||||
* size is attacker-controlled metadata — a crafted ZIP entry can claim
|
||||
* multi-GB sizes that would be allocated up front before any data inflates.
|
||||
*/
|
||||
const MAX_ARCHIVE_MEMBER_BYTES = 64 * 1024 * 1024;
|
||||
|
||||
export type ArchiveFormat = "zip" | "tar" | "tar.gz";
|
||||
|
||||
export interface ArchivePathCandidate {
|
||||
archivePath: string;
|
||||
subPath: string;
|
||||
}
|
||||
|
||||
export interface ArchiveNode {
|
||||
path: string;
|
||||
isDirectory: boolean;
|
||||
size: number;
|
||||
mtimeMs?: number;
|
||||
}
|
||||
|
||||
export interface ArchiveDirectoryEntry extends ArchiveNode {
|
||||
name: string;
|
||||
}
|
||||
|
||||
export interface ExtractedArchiveFile extends ArchiveNode {
|
||||
bytes: Uint8Array;
|
||||
}
|
||||
|
||||
interface TarStorage {
|
||||
type: "tar";
|
||||
file: File;
|
||||
}
|
||||
|
||||
interface ZipStorage {
|
||||
type: "zip";
|
||||
archivePath: string;
|
||||
compressedSize: number;
|
||||
compression: number;
|
||||
flags: number;
|
||||
localHeaderOffset: number;
|
||||
}
|
||||
|
||||
type EntryStorage = TarStorage | ZipStorage;
|
||||
|
||||
interface ArchiveIndexEntry extends ArchiveNode {
|
||||
storage?: EntryStorage;
|
||||
}
|
||||
|
||||
function normalizeArchiveLookupPath(rawPath?: string): string | undefined {
|
||||
if (!rawPath) return "";
|
||||
|
||||
const parts = rawPath.replace(/\\/g, "/").split("/");
|
||||
const normalizedParts: string[] = [];
|
||||
for (const part of parts) {
|
||||
if (!part || part === ".") continue;
|
||||
if (part === "..") return undefined;
|
||||
normalizedParts.push(part);
|
||||
}
|
||||
|
||||
return normalizedParts.join("/");
|
||||
}
|
||||
|
||||
function normalizeArchiveEntryPath(rawPath: string): string | undefined {
|
||||
const parts = rawPath.replace(/\\/g, "/").split("/");
|
||||
const normalizedParts: string[] = [];
|
||||
for (const part of parts) {
|
||||
if (!part || part === ".") continue;
|
||||
if (part === "..") return undefined;
|
||||
normalizedParts.push(part);
|
||||
}
|
||||
|
||||
if (normalizedParts.length === 0) return undefined;
|
||||
return normalizedParts.join("/");
|
||||
}
|
||||
|
||||
function isArchiveDirectoryName(rawPath: string): boolean {
|
||||
return rawPath.endsWith("/") || rawPath.endsWith("\\");
|
||||
}
|
||||
|
||||
function upsertArchiveEntry(map: Map<string, ArchiveIndexEntry>, entry: ArchiveIndexEntry): void {
|
||||
const existing = map.get(entry.path);
|
||||
if (!existing) {
|
||||
map.set(entry.path, entry);
|
||||
return;
|
||||
}
|
||||
|
||||
if (existing.isDirectory && !entry.isDirectory) {
|
||||
map.set(entry.path, entry);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!existing.isDirectory && entry.isDirectory) {
|
||||
return;
|
||||
}
|
||||
|
||||
map.set(entry.path, {
|
||||
...existing,
|
||||
size: existing.size || entry.size,
|
||||
mtimeMs: existing.mtimeMs ?? entry.mtimeMs,
|
||||
storage: existing.storage ?? entry.storage,
|
||||
});
|
||||
}
|
||||
|
||||
function ensureParentDirectories(map: Map<string, ArchiveIndexEntry>): void {
|
||||
for (const entry of [...map.values()]) {
|
||||
const parts = entry.path.split("/");
|
||||
const stop = parts.length - 1;
|
||||
for (let index = 1; index <= stop; index++) {
|
||||
const dirPath = parts.slice(0, index).join("/");
|
||||
if (!dirPath || map.has(dirPath)) continue;
|
||||
map.set(dirPath, {
|
||||
path: dirPath,
|
||||
isDirectory: true,
|
||||
size: 0,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined {
|
||||
const normalized = filePath.toLowerCase();
|
||||
if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz";
|
||||
if (normalized.endsWith(".tar")) return "tar";
|
||||
if (normalized.endsWith(".zip")) return "zip";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] {
|
||||
return entries.map(entry => {
|
||||
if (entry.isDirectory) return `${entry.name}/`;
|
||||
|
||||
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
|
||||
return `${entry.name}${sizeSuffix}`;
|
||||
});
|
||||
}
|
||||
|
||||
const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50;
|
||||
const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50;
|
||||
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
|
||||
const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50;
|
||||
const ZIP_EOCD_SIGNATURE = 0x06054b50;
|
||||
const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50;
|
||||
const ZIP_EOCD_MIN_LENGTH = 22;
|
||||
const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff;
|
||||
const ZIP64_EOCD_LOCATOR_LENGTH = 20;
|
||||
const ZIP_STORED_COMPRESSION = 0;
|
||||
const ZIP_DEFLATE_COMPRESSION = 8;
|
||||
const ZIP_UTF8_FLAG = 0x0800;
|
||||
const ZIP_ENCRYPTED_FLAG = 0x0001;
|
||||
const ZIP_UINT16_MAX = 0xffff;
|
||||
const ZIP_UINT32_MAX = 0xffffffff;
|
||||
const ZIP_UINT32_RANGE = 0x100000000;
|
||||
|
||||
interface ZipCentralDirectoryInfo {
|
||||
entries: number;
|
||||
offset: number;
|
||||
size: number;
|
||||
}
|
||||
|
||||
interface Zip64EntryValues {
|
||||
compressedSize: number;
|
||||
uncompressedSize: number;
|
||||
localHeaderOffset: number;
|
||||
diskStart: number;
|
||||
}
|
||||
|
||||
interface Zip64EntryPlaceholders {
|
||||
compressedSize: boolean;
|
||||
uncompressedSize: boolean;
|
||||
localHeaderOffset: boolean;
|
||||
diskStart: boolean;
|
||||
}
|
||||
|
||||
function readUInt16LE(bytes: Uint8Array, offset: number): number {
|
||||
return bytes[offset]! | (bytes[offset + 1]! << 8);
|
||||
}
|
||||
|
||||
function readUInt32LE(bytes: Uint8Array, offset: number): number {
|
||||
return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0;
|
||||
}
|
||||
|
||||
function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean {
|
||||
if (bytes.byteLength < offset + value.length) return false;
|
||||
for (let index = 0; index < value.length; index++) {
|
||||
if (bytes[offset + index] !== value.charCodeAt(index)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined {
|
||||
if (bytes.byteLength >= 4) {
|
||||
const signature = readUInt32LE(bytes, 0);
|
||||
if (
|
||||
signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE ||
|
||||
signature === ZIP_EOCD_SIGNATURE ||
|
||||
signature === ZIP_DATA_DESCRIPTOR_SIGNATURE
|
||||
) {
|
||||
return "zip";
|
||||
}
|
||||
}
|
||||
|
||||
if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) {
|
||||
return "tar.gz";
|
||||
}
|
||||
|
||||
if (bytesMatchAscii(bytes, 257, "ustar")) {
|
||||
return "tar";
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number {
|
||||
const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE;
|
||||
if (!Number.isSafeInteger(value)) {
|
||||
throw new ToolError("ZIP archive uses offsets or sizes too large to read safely");
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
async function readZipRange(filePath: string, start: number, end: number): Promise<Uint8Array> {
|
||||
if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) {
|
||||
throw new ToolError("Invalid ZIP archive range");
|
||||
}
|
||||
|
||||
const bytes = await Bun.file(filePath).slice(start, end).bytes();
|
||||
if (bytes.byteLength !== end - start) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated data");
|
||||
}
|
||||
return bytes;
|
||||
}
|
||||
|
||||
function findEndOfCentralDirectory(tail: Uint8Array): number {
|
||||
for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) {
|
||||
if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue;
|
||||
const commentLength = readUInt16LE(tail, offset + 20);
|
||||
if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset;
|
||||
}
|
||||
|
||||
throw new ToolError("Invalid ZIP archive: missing end of central directory");
|
||||
}
|
||||
|
||||
async function readZip64CentralDirectoryInfo(
|
||||
filePath: string,
|
||||
tail: Uint8Array,
|
||||
tailStart: number,
|
||||
eocdOffset: number,
|
||||
): Promise<ZipCentralDirectoryInfo | undefined> {
|
||||
const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH;
|
||||
if (locatorOffset < 0) return undefined;
|
||||
|
||||
const locator =
|
||||
locatorOffset >= tailStart
|
||||
? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH)
|
||||
: await readZipRange(filePath, locatorOffset, eocdOffset);
|
||||
if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined;
|
||||
|
||||
const zip64EocdDisk = readUInt32LE(locator, 4);
|
||||
const zip64EocdOffset = readUInt64LEAsNumber(locator, 8);
|
||||
const totalDisks = readUInt32LE(locator, 16);
|
||||
if (zip64EocdDisk !== 0 || totalDisks > 1) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
const record = await readZipRange(filePath, zip64EocdOffset, zip64EocdOffset + 56);
|
||||
if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory");
|
||||
}
|
||||
if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
return {
|
||||
entries: readUInt64LEAsNumber(record, 32),
|
||||
size: readUInt64LEAsNumber(record, 40),
|
||||
offset: readUInt64LEAsNumber(record, 48),
|
||||
};
|
||||
}
|
||||
|
||||
async function readZipCentralDirectoryInfo(filePath: string, fileSize: number): Promise<ZipCentralDirectoryInfo> {
|
||||
if (fileSize < ZIP_EOCD_MIN_LENGTH) {
|
||||
throw new ToolError("Invalid ZIP archive: missing end of central directory");
|
||||
}
|
||||
|
||||
const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH);
|
||||
const tailStart = fileSize - tailLength;
|
||||
const tail = await readZipRange(filePath, tailStart, fileSize);
|
||||
const eocdIndex = findEndOfCentralDirectory(tail);
|
||||
const eocdOffset = tailStart + eocdIndex;
|
||||
|
||||
if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
let entries = readUInt16LE(tail, eocdIndex + 10);
|
||||
let size = readUInt32LE(tail, eocdIndex + 12);
|
||||
let offset = readUInt32LE(tail, eocdIndex + 16);
|
||||
const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX;
|
||||
const zip64Info = await readZip64CentralDirectoryInfo(filePath, tail, tailStart, eocdOffset);
|
||||
if (zip64Info) {
|
||||
({ entries, size, offset } = zip64Info);
|
||||
} else if (needsZip64) {
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata");
|
||||
}
|
||||
|
||||
if (offset + size > fileSize) {
|
||||
throw new ToolError("Invalid ZIP archive: central directory exceeds file size");
|
||||
}
|
||||
|
||||
return { entries, offset, size };
|
||||
}
|
||||
|
||||
function readZip64EntryValues(
|
||||
extra: Uint8Array,
|
||||
placeholders: Zip64EntryPlaceholders,
|
||||
current: Zip64EntryValues,
|
||||
): Zip64EntryValues {
|
||||
if (
|
||||
!placeholders.compressedSize &&
|
||||
!placeholders.uncompressedSize &&
|
||||
!placeholders.localHeaderOffset &&
|
||||
!placeholders.diskStart
|
||||
) {
|
||||
return current;
|
||||
}
|
||||
|
||||
let offset = 0;
|
||||
while (offset + 4 <= extra.byteLength) {
|
||||
const headerId = readUInt16LE(extra, offset);
|
||||
const dataSize = readUInt16LE(extra, offset + 2);
|
||||
const dataStart = offset + 4;
|
||||
const dataEnd = dataStart + dataSize;
|
||||
if (dataEnd > extra.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed extra field");
|
||||
}
|
||||
|
||||
if (headerId === 0x0001) {
|
||||
let cursor = dataStart;
|
||||
let uncompressedSize = current.uncompressedSize;
|
||||
let compressedSize = current.compressedSize;
|
||||
let localHeaderOffset = current.localHeaderOffset;
|
||||
let diskStart = current.diskStart;
|
||||
|
||||
if (placeholders.uncompressedSize) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
uncompressedSize = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.compressedSize) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
compressedSize = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.localHeaderOffset) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
localHeaderOffset = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.diskStart) {
|
||||
if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
diskStart = readUInt32LE(extra, cursor);
|
||||
}
|
||||
|
||||
return { compressedSize, uncompressedSize, localHeaderOffset, diskStart };
|
||||
}
|
||||
|
||||
offset = dataEnd;
|
||||
}
|
||||
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field");
|
||||
}
|
||||
|
||||
function parseZipCentralDirectory(
|
||||
filePath: string,
|
||||
centralDirectory: Uint8Array,
|
||||
expectedEntries: number,
|
||||
): ArchiveIndexEntry[] {
|
||||
const entries: ArchiveIndexEntry[] = [];
|
||||
let offset = 0;
|
||||
|
||||
for (let index = 0; index < expectedEntries; index++) {
|
||||
if (offset + 46 > centralDirectory.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated central directory");
|
||||
}
|
||||
if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed central directory");
|
||||
}
|
||||
|
||||
const flags = readUInt16LE(centralDirectory, offset + 8);
|
||||
const compression = readUInt16LE(centralDirectory, offset + 10);
|
||||
const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20);
|
||||
const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24);
|
||||
const fileNameLength = readUInt16LE(centralDirectory, offset + 28);
|
||||
const extraLength = readUInt16LE(centralDirectory, offset + 30);
|
||||
const commentLength = readUInt16LE(centralDirectory, offset + 32);
|
||||
const diskStartRaw = readUInt16LE(centralDirectory, offset + 34);
|
||||
const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42);
|
||||
const nameStart = offset + 46;
|
||||
const extraStart = nameStart + fileNameLength;
|
||||
const entryEnd = extraStart + extraLength + commentLength;
|
||||
if (entryEnd > centralDirectory.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
|
||||
}
|
||||
|
||||
const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
|
||||
const normalizedPath = normalizeArchiveEntryPath(rawPath);
|
||||
if (normalizedPath) {
|
||||
const values = readZip64EntryValues(
|
||||
centralDirectory.subarray(extraStart, extraStart + extraLength),
|
||||
{
|
||||
compressedSize: compressedSizeRaw === ZIP_UINT32_MAX,
|
||||
uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX,
|
||||
localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX,
|
||||
diskStart: diskStartRaw === ZIP_UINT16_MAX,
|
||||
},
|
||||
{
|
||||
compressedSize: compressedSizeRaw,
|
||||
uncompressedSize: uncompressedSizeRaw,
|
||||
localHeaderOffset: localHeaderOffsetRaw,
|
||||
diskStart: diskStartRaw,
|
||||
},
|
||||
);
|
||||
if (values.diskStart !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
const isDirectory = isArchiveDirectoryName(rawPath);
|
||||
entries.push({
|
||||
path: normalizedPath,
|
||||
isDirectory,
|
||||
size: isDirectory ? 0 : values.uncompressedSize,
|
||||
storage: isDirectory
|
||||
? undefined
|
||||
: {
|
||||
type: "zip",
|
||||
archivePath: filePath,
|
||||
compressedSize: values.compressedSize,
|
||||
compression,
|
||||
flags,
|
||||
localHeaderOffset: values.localHeaderOffset,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
offset = entryEnd;
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise<Uint8Array> {
|
||||
if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) {
|
||||
throw new ToolError("Encrypted ZIP entries are not supported");
|
||||
}
|
||||
|
||||
const localHeader = await readZipRange(
|
||||
storage.archivePath,
|
||||
storage.localHeaderOffset,
|
||||
storage.localHeaderOffset + 30,
|
||||
);
|
||||
if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed local file header");
|
||||
}
|
||||
|
||||
const fileNameLength = readUInt16LE(localHeader, 26);
|
||||
const extraLength = readUInt16LE(localHeader, 28);
|
||||
const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength;
|
||||
const compressedBytes = await readZipRange(storage.archivePath, dataStart, dataStart + storage.compressedSize);
|
||||
|
||||
if (storage.compression === ZIP_STORED_COMPRESSION) {
|
||||
return compressedBytes;
|
||||
}
|
||||
if (storage.compression !== ZIP_DEFLATE_COMPRESSION) {
|
||||
throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`);
|
||||
}
|
||||
|
||||
try {
|
||||
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
}
|
||||
|
||||
async function readTarEntries(bytes: Uint8Array): Promise<ArchiveIndexEntry[]> {
|
||||
let archive: Bun.Archive;
|
||||
try {
|
||||
archive = new Bun.Archive(bytes);
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
let files: Map<string, File>;
|
||||
try {
|
||||
files = await archive.files();
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
const entries: ArchiveIndexEntry[] = [];
|
||||
for (const [rawPath, file] of files) {
|
||||
const normalizedPath = normalizeArchiveEntryPath(rawPath);
|
||||
if (!normalizedPath) continue;
|
||||
const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined;
|
||||
entries.push({
|
||||
path: normalizedPath,
|
||||
isDirectory: false,
|
||||
size: file.size,
|
||||
mtimeMs,
|
||||
storage: { type: "tar", file },
|
||||
});
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
async function readZipEntries(filePath: string): Promise<ArchiveIndexEntry[]> {
|
||||
const fileSize = Bun.file(filePath).size;
|
||||
if (!Number.isSafeInteger(fileSize)) {
|
||||
throw new ToolError("ZIP archive is too large to read safely");
|
||||
}
|
||||
|
||||
const directoryInfo = await readZipCentralDirectoryInfo(filePath, fileSize);
|
||||
const centralDirectory = await readZipRange(
|
||||
filePath,
|
||||
directoryInfo.offset,
|
||||
directoryInfo.offset + directoryInfo.size,
|
||||
);
|
||||
return parseZipCentralDirectory(filePath, centralDirectory, directoryInfo.entries);
|
||||
}
|
||||
|
||||
export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] {
|
||||
const normalized = filePath.replace(/\\/g, "/");
|
||||
const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi;
|
||||
const seen = new Set<string>();
|
||||
const candidates: ArchivePathCandidate[] = [];
|
||||
|
||||
let match: RegExpExecArray | null;
|
||||
while (true) {
|
||||
match = pattern.exec(normalized);
|
||||
if (match === null) {
|
||||
break;
|
||||
}
|
||||
const end = match.index + match[0].length;
|
||||
const archivePath = filePath.slice(0, end);
|
||||
const subPath = normalized.slice(end).replace(/^:+/, "");
|
||||
const key = `${archivePath}\0${subPath}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
candidates.push({ archivePath, subPath });
|
||||
}
|
||||
|
||||
return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length);
|
||||
}
|
||||
|
||||
export class ArchiveReader {
|
||||
readonly format: ArchiveFormat;
|
||||
#entries = new Map<string, ArchiveIndexEntry>();
|
||||
|
||||
constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) {
|
||||
this.format = format;
|
||||
for (const entry of entries) {
|
||||
upsertArchiveEntry(this.#entries, entry);
|
||||
}
|
||||
ensureParentDirectories(this.#entries);
|
||||
}
|
||||
|
||||
getNode(subPath?: string): ArchiveNode | undefined {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (normalizedPath === undefined) return undefined;
|
||||
if (normalizedPath === "") {
|
||||
return { path: "", isDirectory: true, size: 0 };
|
||||
}
|
||||
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) return undefined;
|
||||
return {
|
||||
path: entry.path,
|
||||
isDirectory: entry.isDirectory,
|
||||
size: entry.size,
|
||||
mtimeMs: entry.mtimeMs,
|
||||
};
|
||||
}
|
||||
|
||||
listDirectory(subPath?: string): ArchiveDirectoryEntry[] {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (normalizedPath === undefined) {
|
||||
throw new ToolError("Archive path cannot contain '..'");
|
||||
}
|
||||
|
||||
if (normalizedPath) {
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' not found`);
|
||||
}
|
||||
if (!entry.isDirectory) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' is not a directory`);
|
||||
}
|
||||
}
|
||||
|
||||
const prefix = normalizedPath ? `${normalizedPath}/` : "";
|
||||
const children = new Map<string, ArchiveDirectoryEntry>();
|
||||
|
||||
for (const entry of this.#entries.values()) {
|
||||
if (normalizedPath) {
|
||||
if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue;
|
||||
}
|
||||
|
||||
const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path;
|
||||
const nextSegment = relativePath.split("/")[0];
|
||||
if (!nextSegment) continue;
|
||||
|
||||
const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment;
|
||||
if (children.has(childPath)) continue;
|
||||
|
||||
const childEntry = this.#entries.get(childPath);
|
||||
const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/");
|
||||
children.set(childPath, {
|
||||
name: nextSegment,
|
||||
path: childPath,
|
||||
isDirectory,
|
||||
size: isDirectory ? 0 : (childEntry?.size ?? entry.size),
|
||||
mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs,
|
||||
});
|
||||
}
|
||||
|
||||
return [...children.values()].sort((left, right) =>
|
||||
left.name.toLowerCase().localeCompare(right.name.toLowerCase()),
|
||||
);
|
||||
}
|
||||
|
||||
async readFile(subPath: string): Promise<ExtractedArchiveFile> {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (!normalizedPath) {
|
||||
throw new ToolError("Archive file path is required");
|
||||
}
|
||||
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) {
|
||||
throw new ToolError(`Archive file '${normalizedPath}' not found`);
|
||||
}
|
||||
if (entry.isDirectory) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' is a directory`);
|
||||
}
|
||||
if (!entry.storage) {
|
||||
throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`);
|
||||
}
|
||||
if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) {
|
||||
throw new ToolError(
|
||||
`Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`,
|
||||
);
|
||||
}
|
||||
|
||||
const bytes =
|
||||
entry.storage.type === "tar"
|
||||
? await entry.storage.file.bytes()
|
||||
: await readZipFileBytes(entry.storage, entry.size);
|
||||
|
||||
return {
|
||||
path: entry.path,
|
||||
isDirectory: false,
|
||||
size: entry.size,
|
||||
mtimeMs: entry.mtimeMs,
|
||||
bytes,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export async function openArchive(filePath: string): Promise<ArchiveReader> {
|
||||
const format = getArchiveFormatFromPath(filePath);
|
||||
if (!format) {
|
||||
throw new ToolError(`Unsupported archive format: ${filePath}`);
|
||||
}
|
||||
|
||||
if (format === "zip") {
|
||||
return new ArchiveReader(format, await readZipEntries(filePath));
|
||||
}
|
||||
|
||||
const file = Bun.file(filePath);
|
||||
const archiveSize = file.size;
|
||||
if (archiveSize > MAX_TAR_ARCHIVE_BYTES) {
|
||||
throw new ToolError(
|
||||
`Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
|
||||
);
|
||||
}
|
||||
const entries = await readTarEntries(await file.bytes());
|
||||
return new ArchiveReader(format, entries);
|
||||
}
|
||||
|
||||
export async function listArchiveRoot(
|
||||
bytes: Uint8Array,
|
||||
format: ArchiveFormat,
|
||||
opts: { limit?: number } = {},
|
||||
): Promise<string> {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-archive-"));
|
||||
const tempPath = path.join(tempDir, `payload.${format}`);
|
||||
try {
|
||||
await Bun.write(tempPath, bytes);
|
||||
const archive = await openArchive(tempPath);
|
||||
const entries = archive.listDirectory("");
|
||||
const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries;
|
||||
const lines = formatArchiveEntryLines(limitedEntries);
|
||||
return lines.length > 0 ? lines.join("\n") : "(empty archive directory)";
|
||||
} finally {
|
||||
await fs.rm(tempDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
@@ -20,12 +20,12 @@ import { CachedOutputBlock, markFramedBlockComponent } from "../tui/output-block
|
||||
import { webpExclusionForModel } from "../utils/image-loading";
|
||||
import { formatDimensionNote, resizeImage } from "../utils/image-resize";
|
||||
import { ensureTool } from "../utils/tools-manager";
|
||||
import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "../utils/zip";
|
||||
import { extractWithParallel, findParallelApiKey, getParallelExtractContent } from "../web/parallel";
|
||||
import { specialHandlers } from "../web/scrapers";
|
||||
import type { RenderResult } from "../web/scrapers/types";
|
||||
import { finalizeOutput, loadPage, looksLikeHtml, MAX_BYTES, MAX_OUTPUT_CHARS } from "../web/scrapers/types";
|
||||
import { convertWithMarkit, fetchBinary } from "../web/scrapers/utils";
|
||||
import { type ArchiveFormat, listArchiveRoot, sniffArchiveFormat } from "./archive-reader";
|
||||
import { applyListLimit } from "./list-limit";
|
||||
import { formatStyledArtifactReference, type OutputMeta } from "./output-meta";
|
||||
import { type LineRange, parseLineRanges } from "./path-utils";
|
||||
|
||||
@@ -48,8 +48,8 @@ import {
|
||||
webpExclusionForModel,
|
||||
} from "../utils/image-loading";
|
||||
import { convertFileWithMarkit } from "../utils/markit";
|
||||
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "../utils/zip";
|
||||
import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree";
|
||||
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader";
|
||||
import {
|
||||
type ConflictEntry,
|
||||
type ConflictScope,
|
||||
|
||||
@@ -28,13 +28,8 @@ import {
|
||||
uriHyperlink,
|
||||
} from "../tui";
|
||||
import { resolveFileDisplayMode } from "../utils/file-display-mode";
|
||||
import { type ArchiveReader, type ExtractedArchiveFile, openArchive, parseArchivePathCandidates } from "../utils/zip";
|
||||
import type { ToolSession } from ".";
|
||||
import {
|
||||
type ArchiveReader,
|
||||
type ExtractedArchiveFile,
|
||||
openArchive,
|
||||
parseArchivePathCandidates,
|
||||
} from "./archive-reader";
|
||||
import { createFileRecorder, formatResultPath } from "./file-recorder";
|
||||
import { classifyGroupedLines, formatGroupedFiles, groupLineIndicesByBlank } from "./grouped-file-output";
|
||||
import { formatMatchLine } from "./match-line-format";
|
||||
|
||||
@@ -20,8 +20,14 @@ import writeDescription from "../prompts/tools/write.md" with { type: "text" };
|
||||
import type { ToolSession } from "../sdk";
|
||||
import { fileHyperlink, framedBlock, renderStatusLine } from "../tui";
|
||||
import { resolveFileDisplayMode } from "../utils/file-display-mode";
|
||||
import {
|
||||
type ArchiveMemberContent,
|
||||
archiveFormatFromPath,
|
||||
parseArchivePathCandidates,
|
||||
readArchiveEntries,
|
||||
writeArchive,
|
||||
} from "../utils/zip";
|
||||
import { truncateForPrompt } from "./approval";
|
||||
import { parseArchivePathCandidates } from "./archive-reader";
|
||||
import { assertEditableFile } from "./auto-generated-guard";
|
||||
import {
|
||||
type ConflictEntry,
|
||||
@@ -363,9 +369,10 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
|
||||
const finalPath = resolvedArchivePath.exists
|
||||
? await fs.realpath(resolvedArchivePath.absolutePath).catch(() => resolvedArchivePath.absolutePath)
|
||||
: resolvedArchivePath.absolutePath;
|
||||
const lowerPath = finalPath.toLowerCase();
|
||||
const isZip = lowerPath.endsWith(".zip");
|
||||
const isGzip = lowerPath.endsWith(".tar.gz") || lowerPath.endsWith(".tgz");
|
||||
// A realpath swap can land on a name without an archive extension; a
|
||||
// whole-archive rewrite then defaults to an uncompressed tar, matching the
|
||||
// previous `isZip`/`isGzip`/else fallthrough.
|
||||
const format = archiveFormatFromPath(finalPath) ?? "tar";
|
||||
// Rewrites are whole-archive: write to a temp file and rename so a
|
||||
// crash/disk-full mid-write can't destroy the original archive.
|
||||
const tmpPath = `${finalPath}.tmp-${process.pid}`;
|
||||
@@ -375,66 +382,25 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
|
||||
await fs.mkdir(parentDir, { recursive: true });
|
||||
}
|
||||
|
||||
if (isZip) {
|
||||
const zipEntries: Record<string, Uint8Array> = {};
|
||||
|
||||
if (resolvedArchivePath.exists) {
|
||||
try {
|
||||
const bytes = await Bun.file(resolvedArchivePath.absolutePath).bytes();
|
||||
const { unzip } = await import("../utils/zip");
|
||||
const existing = unzip(new Uint8Array(bytes));
|
||||
for (const [entryPath, data] of Object.entries(existing)) {
|
||||
zipEntries[entryPath.replace(/\\/g, "/")] = data;
|
||||
}
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
}
|
||||
|
||||
zipEntries[resolvedArchivePath.archiveSubPath] = new TextEncoder().encode(content);
|
||||
|
||||
const entries = new Map<string, ArchiveMemberContent>();
|
||||
if (resolvedArchivePath.exists) {
|
||||
try {
|
||||
const { zip } = await import("../utils/zip");
|
||||
const zipBuffer = zip(zipEntries);
|
||||
await Bun.write(tmpPath, zipBuffer);
|
||||
await fs.rename(tmpPath, finalPath);
|
||||
const existing = await readArchiveEntries({ bytes: await Bun.file(finalPath).bytes(), format });
|
||||
for (const [entryPath, data] of existing) {
|
||||
entries.set(entryPath, data);
|
||||
}
|
||||
} catch (error) {
|
||||
await fs.rm(tmpPath, { force: true }).catch(() => {});
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
} else {
|
||||
const archiveEntries: Record<string, string | File> = {};
|
||||
if (resolvedArchivePath.exists) {
|
||||
let archive: Bun.Archive;
|
||||
try {
|
||||
archive = new Bun.Archive(await Bun.file(resolvedArchivePath.absolutePath).bytes());
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
}
|
||||
entries.set(resolvedArchivePath.archiveSubPath, content);
|
||||
|
||||
let files: Map<string, File>;
|
||||
try {
|
||||
files = await archive.files();
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
for (const [entryPath, file] of files) {
|
||||
archiveEntries[entryPath.replace(/\\/g, "/")] = file;
|
||||
}
|
||||
}
|
||||
|
||||
archiveEntries[resolvedArchivePath.archiveSubPath] = content;
|
||||
|
||||
try {
|
||||
// `Bun.Archive.write` never infers compression from the extension;
|
||||
// request gzip explicitly so `.tar.gz`/`.tgz` stay compressed.
|
||||
await Bun.Archive.write(tmpPath, archiveEntries, isGzip ? { compress: "gzip" } : undefined);
|
||||
await fs.rename(tmpPath, finalPath);
|
||||
} catch (error) {
|
||||
await fs.rm(tmpPath, { force: true }).catch(() => {});
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
try {
|
||||
await writeArchive(tmpPath, format, entries);
|
||||
await fs.rename(tmpPath, finalPath);
|
||||
} catch (error) {
|
||||
await fs.rm(tmpPath, { force: true }).catch(() => {});
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
invalidateFsScanAfterWrite(resolvedArchivePath.absolutePath);
|
||||
|
||||
@@ -2,6 +2,7 @@ import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { $which, APP_NAME, getToolsDir, logger, ptree, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { extractArchive } from "./zip";
|
||||
|
||||
const TOOLS_DIR = getToolsDir();
|
||||
const TOOL_DOWNLOAD_TIMEOUT_MS = 120_000;
|
||||
@@ -220,17 +221,7 @@ async function downloadTool(tool: ToolName, signal?: AbortSignal): Promise<strin
|
||||
}
|
||||
|
||||
try {
|
||||
const archive = new Bun.Archive(await Bun.file(archivePath).arrayBuffer());
|
||||
const files = await archive.files();
|
||||
const extractRoot = path.resolve(tmp.path());
|
||||
|
||||
for (const [filePath, file] of files) {
|
||||
const outputPath = path.resolve(extractRoot, filePath);
|
||||
if (!outputPath.startsWith(extractRoot + path.sep)) {
|
||||
throw new Error(`Archive entry escapes extraction dir: ${filePath}`);
|
||||
}
|
||||
await Bun.write(outputPath, file);
|
||||
}
|
||||
await extractArchive(archivePath, tmp.path());
|
||||
} catch (err) {
|
||||
throw new Error(`Failed to extract ${assetName}: ${err instanceof Error ? err.message : String(err)}`);
|
||||
}
|
||||
|
||||
@@ -1,29 +1,891 @@
|
||||
// The single ZIP/DEFLATE boundary for the codebase. This is the ONLY module
|
||||
// that imports `fflate`; the markit document converters, the write tool, and
|
||||
// the archive reader all go through here so there is exactly one ZIP
|
||||
// implementation to reason about. Do not import `fflate` (or another archive
|
||||
// library) anywhere else.
|
||||
// The single archive boundary for the codebase: ZIP (via `fflate`) and tar /
|
||||
// tar.gz (via `Bun.Archive`), built on the raw DEFLATE helpers the ZIP reader
|
||||
// needs. This is the ONLY module that imports `fflate` or touches
|
||||
// `Bun.Archive`; the markit document converters, the read/search/write tools,
|
||||
// the URL fetcher, the debug report bundler, and the tool-binary installer all
|
||||
// go through here so there is exactly one archive implementation to reason
|
||||
// about. Do not import `fflate` (or call `Bun.Archive`) anywhere else.
|
||||
import * as path from "node:path";
|
||||
import { formatBytes } from "@oh-my-pi/pi-utils";
|
||||
import type { Unzipped } from "fflate";
|
||||
import { inflateSync, strFromU8 } from "fflate";
|
||||
import { inflateSync, unzipSync, zipSync } from "fflate";
|
||||
import { ToolError } from "../tools/tool-errors";
|
||||
|
||||
export type { Unzipped } from "fflate";
|
||||
export { unzipSync as unzip, zipSync as zip } from "fflate";
|
||||
|
||||
const ENCODER = new TextEncoder();
|
||||
// fflate's `strFromU8` is just `new TextDecoder().decode()` for UTF-8, so the
|
||||
// converters and ZIP filename reader use the platform decoders directly.
|
||||
const UTF8_DECODER = new TextDecoder();
|
||||
// ZIP central-directory names without the UTF-8 flag carry no reliable encoding;
|
||||
// decode them as their legacy code page (windows-1252) as a stable best effort.
|
||||
const LEGACY_NAME_DECODER = new TextDecoder("windows-1252");
|
||||
|
||||
/** Read a single ZIP entry as UTF-8 text, or `undefined` when the entry is absent. */
|
||||
export function unzipText(entries: Unzipped, entryPath: string): string | undefined {
|
||||
const data = entries[entryPath];
|
||||
return data ? strFromU8(data) : undefined;
|
||||
return data ? UTF8_DECODER.decode(data) : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Inflate a raw DEFLATE stream (a single deflate-compressed ZIP member). Pass a
|
||||
* preallocated `into` buffer when the uncompressed size is known up front.
|
||||
*/
|
||||
export function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
|
||||
function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
|
||||
return into ? inflateSync(bytes, { out: into }) : inflateSync(bytes);
|
||||
}
|
||||
|
||||
/** Decode raw bytes as text — UTF-8 by default, latin1 when `latin1` is set. */
|
||||
export function bytesToText(bytes: Uint8Array, latin1?: boolean): string {
|
||||
return strFromU8(bytes, latin1);
|
||||
/**
|
||||
* Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into
|
||||
* memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is
|
||||
* exempt: it is read via ranged central-directory access.
|
||||
*/
|
||||
const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024;
|
||||
/**
|
||||
* Cap on a single archive member's declared (uncompressed) size. The declared
|
||||
* size is attacker-controlled metadata — a crafted ZIP entry can claim
|
||||
* multi-GB sizes that would be allocated up front before any data inflates.
|
||||
*/
|
||||
const MAX_ARCHIVE_MEMBER_BYTES = 64 * 1024 * 1024;
|
||||
|
||||
export type ArchiveFormat = "zip" | "tar" | "tar.gz";
|
||||
|
||||
/**
|
||||
* Where to read an archive from: a filesystem path (format inferred from the
|
||||
* extension; ZIP is read lazily via ranged central-directory access) or
|
||||
* in-memory bytes with an explicit format.
|
||||
*/
|
||||
export type ArchiveSource = string | { bytes: Uint8Array; format: ArchiveFormat };
|
||||
|
||||
/** Content for a member when packing or extracting an archive. */
|
||||
export type ArchiveMemberContent = string | Uint8Array | Blob;
|
||||
|
||||
export interface ArchivePathCandidate {
|
||||
archivePath: string;
|
||||
subPath: string;
|
||||
}
|
||||
|
||||
export interface ArchiveNode {
|
||||
path: string;
|
||||
isDirectory: boolean;
|
||||
size: number;
|
||||
mtimeMs?: number;
|
||||
}
|
||||
|
||||
export interface ArchiveDirectoryEntry extends ArchiveNode {
|
||||
name: string;
|
||||
}
|
||||
|
||||
export interface ExtractedArchiveFile extends ArchiveNode {
|
||||
bytes: Uint8Array;
|
||||
}
|
||||
|
||||
/** A byte window into an archive — file-backed (lazy) or in-memory. */
|
||||
interface ByteSource {
|
||||
readonly size: number;
|
||||
read(start: number, end: number): Promise<Uint8Array>;
|
||||
}
|
||||
|
||||
function assertValidRange(start: number, end: number): void {
|
||||
if (!Number.isSafeInteger(start) || !Number.isSafeInteger(end) || start < 0 || end < start) {
|
||||
throw new ToolError("Invalid ZIP archive range");
|
||||
}
|
||||
}
|
||||
|
||||
function fileByteSource(filePath: string): ByteSource {
|
||||
const file = Bun.file(filePath);
|
||||
const size = file.size;
|
||||
if (!Number.isSafeInteger(size)) {
|
||||
throw new ToolError("ZIP archive is too large to read safely");
|
||||
}
|
||||
return {
|
||||
size,
|
||||
async read(start, end) {
|
||||
assertValidRange(start, end);
|
||||
const bytes = await file.slice(start, end).bytes();
|
||||
if (bytes.byteLength !== end - start) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated data");
|
||||
}
|
||||
return bytes;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function memoryByteSource(buffer: Uint8Array): ByteSource {
|
||||
return {
|
||||
size: buffer.byteLength,
|
||||
async read(start, end) {
|
||||
assertValidRange(start, end);
|
||||
if (end > buffer.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated data");
|
||||
}
|
||||
return buffer.subarray(start, end);
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
interface TarStorage {
|
||||
type: "tar";
|
||||
file: File;
|
||||
}
|
||||
|
||||
interface ZipStorage {
|
||||
type: "zip";
|
||||
source: ByteSource;
|
||||
compressedSize: number;
|
||||
compression: number;
|
||||
flags: number;
|
||||
localHeaderOffset: number;
|
||||
}
|
||||
|
||||
type EntryStorage = TarStorage | ZipStorage;
|
||||
|
||||
interface ArchiveIndexEntry extends ArchiveNode {
|
||||
storage?: EntryStorage;
|
||||
}
|
||||
|
||||
function normalizeArchiveLookupPath(rawPath?: string): string | undefined {
|
||||
if (!rawPath) return "";
|
||||
|
||||
const parts = rawPath.replace(/\\/g, "/").split("/");
|
||||
const normalizedParts: string[] = [];
|
||||
for (const part of parts) {
|
||||
if (!part || part === ".") continue;
|
||||
if (part === "..") return undefined;
|
||||
normalizedParts.push(part);
|
||||
}
|
||||
|
||||
return normalizedParts.join("/");
|
||||
}
|
||||
|
||||
function normalizeArchiveEntryPath(rawPath: string): string | undefined {
|
||||
const parts = rawPath.replace(/\\/g, "/").split("/");
|
||||
const normalizedParts: string[] = [];
|
||||
for (const part of parts) {
|
||||
if (!part || part === ".") continue;
|
||||
if (part === "..") return undefined;
|
||||
normalizedParts.push(part);
|
||||
}
|
||||
|
||||
if (normalizedParts.length === 0) return undefined;
|
||||
return normalizedParts.join("/");
|
||||
}
|
||||
|
||||
function isArchiveDirectoryName(rawPath: string): boolean {
|
||||
return rawPath.endsWith("/") || rawPath.endsWith("\\");
|
||||
}
|
||||
|
||||
function upsertArchiveEntry(map: Map<string, ArchiveIndexEntry>, entry: ArchiveIndexEntry): void {
|
||||
const existing = map.get(entry.path);
|
||||
if (!existing) {
|
||||
map.set(entry.path, entry);
|
||||
return;
|
||||
}
|
||||
|
||||
if (existing.isDirectory && !entry.isDirectory) {
|
||||
map.set(entry.path, entry);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!existing.isDirectory && entry.isDirectory) {
|
||||
return;
|
||||
}
|
||||
|
||||
map.set(entry.path, {
|
||||
...existing,
|
||||
size: existing.size || entry.size,
|
||||
mtimeMs: existing.mtimeMs ?? entry.mtimeMs,
|
||||
storage: existing.storage ?? entry.storage,
|
||||
});
|
||||
}
|
||||
|
||||
function ensureParentDirectories(map: Map<string, ArchiveIndexEntry>): void {
|
||||
for (const entry of [...map.values()]) {
|
||||
const parts = entry.path.split("/");
|
||||
const stop = parts.length - 1;
|
||||
for (let index = 1; index <= stop; index++) {
|
||||
const dirPath = parts.slice(0, index).join("/");
|
||||
if (!dirPath || map.has(dirPath)) continue;
|
||||
map.set(dirPath, {
|
||||
path: dirPath,
|
||||
isDirectory: true,
|
||||
size: 0,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Infer an archive format from a filesystem path's extension. */
|
||||
export function archiveFormatFromPath(filePath: string): ArchiveFormat | undefined {
|
||||
const normalized = filePath.toLowerCase();
|
||||
if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz";
|
||||
if (normalized.endsWith(".tar")) return "tar";
|
||||
if (normalized.endsWith(".zip")) return "zip";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function formatArchiveEntryLines(entries: readonly ArchiveDirectoryEntry[]): string[] {
|
||||
return entries.map(entry => {
|
||||
if (entry.isDirectory) return `${entry.name}/`;
|
||||
|
||||
const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : "";
|
||||
return `${entry.name}${sizeSuffix}`;
|
||||
});
|
||||
}
|
||||
|
||||
const ZIP_LOCAL_FILE_HEADER_SIGNATURE = 0x04034b50;
|
||||
const ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE = 0x02014b50;
|
||||
const ZIP64_EOCD_SIGNATURE = 0x06064b50;
|
||||
const ZIP64_EOCD_LOCATOR_SIGNATURE = 0x07064b50;
|
||||
const ZIP_EOCD_SIGNATURE = 0x06054b50;
|
||||
const ZIP_DATA_DESCRIPTOR_SIGNATURE = 0x08074b50;
|
||||
const ZIP_EOCD_MIN_LENGTH = 22;
|
||||
const ZIP_EOCD_MAX_COMMENT_LENGTH = 0xffff;
|
||||
const ZIP64_EOCD_LOCATOR_LENGTH = 20;
|
||||
const ZIP_STORED_COMPRESSION = 0;
|
||||
const ZIP_DEFLATE_COMPRESSION = 8;
|
||||
const ZIP_UTF8_FLAG = 0x0800;
|
||||
const ZIP_ENCRYPTED_FLAG = 0x0001;
|
||||
const ZIP_UINT16_MAX = 0xffff;
|
||||
const ZIP_UINT32_MAX = 0xffffffff;
|
||||
const ZIP_UINT32_RANGE = 0x100000000;
|
||||
|
||||
interface ZipCentralDirectoryInfo {
|
||||
entries: number;
|
||||
offset: number;
|
||||
size: number;
|
||||
}
|
||||
|
||||
interface Zip64EntryValues {
|
||||
compressedSize: number;
|
||||
uncompressedSize: number;
|
||||
localHeaderOffset: number;
|
||||
diskStart: number;
|
||||
}
|
||||
|
||||
interface Zip64EntryPlaceholders {
|
||||
compressedSize: boolean;
|
||||
uncompressedSize: boolean;
|
||||
localHeaderOffset: boolean;
|
||||
diskStart: boolean;
|
||||
}
|
||||
|
||||
function readUInt16LE(bytes: Uint8Array, offset: number): number {
|
||||
return bytes[offset]! | (bytes[offset + 1]! << 8);
|
||||
}
|
||||
|
||||
function readUInt32LE(bytes: Uint8Array, offset: number): number {
|
||||
return (bytes[offset]! | (bytes[offset + 1]! << 8) | (bytes[offset + 2]! << 16) | (bytes[offset + 3]! << 24)) >>> 0;
|
||||
}
|
||||
|
||||
function bytesMatchAscii(bytes: Uint8Array, offset: number, value: string): boolean {
|
||||
if (bytes.byteLength < offset + value.length) return false;
|
||||
for (let index = 0; index < value.length; index++) {
|
||||
if (bytes[offset + index] !== value.charCodeAt(index)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
export function sniffArchiveFormat(bytes: Uint8Array): ArchiveFormat | undefined {
|
||||
if (bytes.byteLength >= 4) {
|
||||
const signature = readUInt32LE(bytes, 0);
|
||||
if (
|
||||
signature === ZIP_LOCAL_FILE_HEADER_SIGNATURE ||
|
||||
signature === ZIP_EOCD_SIGNATURE ||
|
||||
signature === ZIP_DATA_DESCRIPTOR_SIGNATURE
|
||||
) {
|
||||
return "zip";
|
||||
}
|
||||
}
|
||||
|
||||
if (bytes.byteLength >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b) {
|
||||
return "tar.gz";
|
||||
}
|
||||
|
||||
if (bytesMatchAscii(bytes, 257, "ustar")) {
|
||||
return "tar";
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function readUInt64LEAsNumber(bytes: Uint8Array, offset: number): number {
|
||||
const value = readUInt32LE(bytes, offset) + readUInt32LE(bytes, offset + 4) * ZIP_UINT32_RANGE;
|
||||
if (!Number.isSafeInteger(value)) {
|
||||
throw new ToolError("ZIP archive uses offsets or sizes too large to read safely");
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function findEndOfCentralDirectory(tail: Uint8Array): number {
|
||||
for (let offset = tail.byteLength - ZIP_EOCD_MIN_LENGTH; offset >= 0; offset--) {
|
||||
if (readUInt32LE(tail, offset) !== ZIP_EOCD_SIGNATURE) continue;
|
||||
const commentLength = readUInt16LE(tail, offset + 20);
|
||||
if (offset + ZIP_EOCD_MIN_LENGTH + commentLength === tail.byteLength) return offset;
|
||||
}
|
||||
|
||||
throw new ToolError("Invalid ZIP archive: missing end of central directory");
|
||||
}
|
||||
|
||||
async function readZip64CentralDirectoryInfo(
|
||||
source: ByteSource,
|
||||
tail: Uint8Array,
|
||||
tailStart: number,
|
||||
eocdOffset: number,
|
||||
): Promise<ZipCentralDirectoryInfo | undefined> {
|
||||
const locatorOffset = eocdOffset - ZIP64_EOCD_LOCATOR_LENGTH;
|
||||
if (locatorOffset < 0) return undefined;
|
||||
|
||||
const locator =
|
||||
locatorOffset >= tailStart
|
||||
? tail.subarray(locatorOffset - tailStart, locatorOffset - tailStart + ZIP64_EOCD_LOCATOR_LENGTH)
|
||||
: await source.read(locatorOffset, eocdOffset);
|
||||
if (readUInt32LE(locator, 0) !== ZIP64_EOCD_LOCATOR_SIGNATURE) return undefined;
|
||||
|
||||
const zip64EocdDisk = readUInt32LE(locator, 4);
|
||||
const zip64EocdOffset = readUInt64LEAsNumber(locator, 8);
|
||||
const totalDisks = readUInt32LE(locator, 16);
|
||||
if (zip64EocdDisk !== 0 || totalDisks > 1) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
const record = await source.read(zip64EocdOffset, zip64EocdOffset + 56);
|
||||
if (readUInt32LE(record, 0) !== ZIP64_EOCD_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 end of central directory");
|
||||
}
|
||||
if (readUInt32LE(record, 16) !== 0 || readUInt32LE(record, 20) !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
return {
|
||||
entries: readUInt64LEAsNumber(record, 32),
|
||||
size: readUInt64LEAsNumber(record, 40),
|
||||
offset: readUInt64LEAsNumber(record, 48),
|
||||
};
|
||||
}
|
||||
|
||||
async function readZipCentralDirectoryInfo(source: ByteSource): Promise<ZipCentralDirectoryInfo> {
|
||||
const fileSize = source.size;
|
||||
if (fileSize < ZIP_EOCD_MIN_LENGTH) {
|
||||
throw new ToolError("Invalid ZIP archive: missing end of central directory");
|
||||
}
|
||||
|
||||
const tailLength = Math.min(fileSize, ZIP_EOCD_MIN_LENGTH + ZIP_EOCD_MAX_COMMENT_LENGTH);
|
||||
const tailStart = fileSize - tailLength;
|
||||
const tail = await source.read(tailStart, fileSize);
|
||||
const eocdIndex = findEndOfCentralDirectory(tail);
|
||||
const eocdOffset = tailStart + eocdIndex;
|
||||
|
||||
if (readUInt16LE(tail, eocdIndex + 4) !== 0 || readUInt16LE(tail, eocdIndex + 6) !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
let entries = readUInt16LE(tail, eocdIndex + 10);
|
||||
let size = readUInt32LE(tail, eocdIndex + 12);
|
||||
let offset = readUInt32LE(tail, eocdIndex + 16);
|
||||
const needsZip64 = entries === ZIP_UINT16_MAX || size === ZIP_UINT32_MAX || offset === ZIP_UINT32_MAX;
|
||||
const zip64Info = await readZip64CentralDirectoryInfo(source, tail, tailStart, eocdOffset);
|
||||
if (zip64Info) {
|
||||
({ entries, size, offset } = zip64Info);
|
||||
} else if (needsZip64) {
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 central directory metadata");
|
||||
}
|
||||
|
||||
if (offset + size > fileSize) {
|
||||
throw new ToolError("Invalid ZIP archive: central directory exceeds file size");
|
||||
}
|
||||
|
||||
return { entries, offset, size };
|
||||
}
|
||||
|
||||
function readZip64EntryValues(
|
||||
extra: Uint8Array,
|
||||
placeholders: Zip64EntryPlaceholders,
|
||||
current: Zip64EntryValues,
|
||||
): Zip64EntryValues {
|
||||
if (
|
||||
!placeholders.compressedSize &&
|
||||
!placeholders.uncompressedSize &&
|
||||
!placeholders.localHeaderOffset &&
|
||||
!placeholders.diskStart
|
||||
) {
|
||||
return current;
|
||||
}
|
||||
|
||||
let offset = 0;
|
||||
while (offset + 4 <= extra.byteLength) {
|
||||
const headerId = readUInt16LE(extra, offset);
|
||||
const dataSize = readUInt16LE(extra, offset + 2);
|
||||
const dataStart = offset + 4;
|
||||
const dataEnd = dataStart + dataSize;
|
||||
if (dataEnd > extra.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed extra field");
|
||||
}
|
||||
|
||||
if (headerId === 0x0001) {
|
||||
let cursor = dataStart;
|
||||
let uncompressedSize = current.uncompressedSize;
|
||||
let compressedSize = current.compressedSize;
|
||||
let localHeaderOffset = current.localHeaderOffset;
|
||||
let diskStart = current.diskStart;
|
||||
|
||||
if (placeholders.uncompressedSize) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
uncompressedSize = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.compressedSize) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
compressedSize = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.localHeaderOffset) {
|
||||
if (cursor + 8 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
localHeaderOffset = readUInt64LEAsNumber(extra, cursor);
|
||||
cursor += 8;
|
||||
}
|
||||
if (placeholders.diskStart) {
|
||||
if (cursor + 4 > dataEnd) throw new ToolError("Invalid ZIP archive: malformed ZIP64 extra field");
|
||||
diskStart = readUInt32LE(extra, cursor);
|
||||
}
|
||||
|
||||
return { compressedSize, uncompressedSize, localHeaderOffset, diskStart };
|
||||
}
|
||||
|
||||
offset = dataEnd;
|
||||
}
|
||||
|
||||
throw new ToolError("Invalid ZIP archive: missing ZIP64 extra field");
|
||||
}
|
||||
|
||||
function parseZipCentralDirectory(
|
||||
source: ByteSource,
|
||||
centralDirectory: Uint8Array,
|
||||
expectedEntries: number,
|
||||
): ArchiveIndexEntry[] {
|
||||
const entries: ArchiveIndexEntry[] = [];
|
||||
let offset = 0;
|
||||
|
||||
for (let index = 0; index < expectedEntries; index++) {
|
||||
if (offset + 46 > centralDirectory.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated central directory");
|
||||
}
|
||||
if (readUInt32LE(centralDirectory, offset) !== ZIP_CENTRAL_DIRECTORY_HEADER_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed central directory");
|
||||
}
|
||||
|
||||
const flags = readUInt16LE(centralDirectory, offset + 8);
|
||||
const compression = readUInt16LE(centralDirectory, offset + 10);
|
||||
const compressedSizeRaw = readUInt32LE(centralDirectory, offset + 20);
|
||||
const uncompressedSizeRaw = readUInt32LE(centralDirectory, offset + 24);
|
||||
const fileNameLength = readUInt16LE(centralDirectory, offset + 28);
|
||||
const extraLength = readUInt16LE(centralDirectory, offset + 30);
|
||||
const commentLength = readUInt16LE(centralDirectory, offset + 32);
|
||||
const diskStartRaw = readUInt16LE(centralDirectory, offset + 34);
|
||||
const localHeaderOffsetRaw = readUInt32LE(centralDirectory, offset + 42);
|
||||
const nameStart = offset + 46;
|
||||
const extraStart = nameStart + fileNameLength;
|
||||
const entryEnd = extraStart + extraLength + commentLength;
|
||||
if (entryEnd > centralDirectory.byteLength) {
|
||||
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
|
||||
}
|
||||
|
||||
const useLegacyEncoding = (flags & ZIP_UTF8_FLAG) === 0;
|
||||
const rawPath = (useLegacyEncoding ? LEGACY_NAME_DECODER : UTF8_DECODER).decode(
|
||||
centralDirectory.subarray(nameStart, extraStart),
|
||||
);
|
||||
const normalizedPath = normalizeArchiveEntryPath(rawPath);
|
||||
if (normalizedPath) {
|
||||
const values = readZip64EntryValues(
|
||||
centralDirectory.subarray(extraStart, extraStart + extraLength),
|
||||
{
|
||||
compressedSize: compressedSizeRaw === ZIP_UINT32_MAX,
|
||||
uncompressedSize: uncompressedSizeRaw === ZIP_UINT32_MAX,
|
||||
localHeaderOffset: localHeaderOffsetRaw === ZIP_UINT32_MAX,
|
||||
diskStart: diskStartRaw === ZIP_UINT16_MAX,
|
||||
},
|
||||
{
|
||||
compressedSize: compressedSizeRaw,
|
||||
uncompressedSize: uncompressedSizeRaw,
|
||||
localHeaderOffset: localHeaderOffsetRaw,
|
||||
diskStart: diskStartRaw,
|
||||
},
|
||||
);
|
||||
if (values.diskStart !== 0) {
|
||||
throw new ToolError("Multi-disk ZIP archives are not supported");
|
||||
}
|
||||
|
||||
const isDirectory = isArchiveDirectoryName(rawPath);
|
||||
entries.push({
|
||||
path: normalizedPath,
|
||||
isDirectory,
|
||||
size: isDirectory ? 0 : values.uncompressedSize,
|
||||
storage: isDirectory
|
||||
? undefined
|
||||
: {
|
||||
type: "zip",
|
||||
source,
|
||||
compressedSize: values.compressedSize,
|
||||
compression,
|
||||
flags,
|
||||
localHeaderOffset: values.localHeaderOffset,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
offset = entryEnd;
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): Promise<Uint8Array> {
|
||||
if ((storage.flags & ZIP_ENCRYPTED_FLAG) !== 0) {
|
||||
throw new ToolError("Encrypted ZIP entries are not supported");
|
||||
}
|
||||
|
||||
const localHeader = await storage.source.read(storage.localHeaderOffset, storage.localHeaderOffset + 30);
|
||||
if (readUInt32LE(localHeader, 0) !== ZIP_LOCAL_FILE_HEADER_SIGNATURE) {
|
||||
throw new ToolError("Invalid ZIP archive: malformed local file header");
|
||||
}
|
||||
|
||||
const fileNameLength = readUInt16LE(localHeader, 26);
|
||||
const extraLength = readUInt16LE(localHeader, 28);
|
||||
const dataStart = storage.localHeaderOffset + 30 + fileNameLength + extraLength;
|
||||
const compressedBytes = await storage.source.read(dataStart, dataStart + storage.compressedSize);
|
||||
|
||||
if (storage.compression === ZIP_STORED_COMPRESSION) {
|
||||
return compressedBytes;
|
||||
}
|
||||
if (storage.compression !== ZIP_DEFLATE_COMPRESSION) {
|
||||
throw new ToolError(`Unsupported ZIP compression method: ${storage.compression}`);
|
||||
}
|
||||
|
||||
try {
|
||||
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
}
|
||||
|
||||
async function readTarEntries(bytes: Uint8Array): Promise<ArchiveIndexEntry[]> {
|
||||
let archive: Bun.Archive;
|
||||
try {
|
||||
archive = new Bun.Archive(bytes);
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
let files: Map<string, File>;
|
||||
try {
|
||||
files = await archive.files();
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
const entries: ArchiveIndexEntry[] = [];
|
||||
for (const [rawPath, file] of files) {
|
||||
const normalizedPath = normalizeArchiveEntryPath(rawPath);
|
||||
if (!normalizedPath) continue;
|
||||
const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined;
|
||||
entries.push({
|
||||
path: normalizedPath,
|
||||
isDirectory: false,
|
||||
size: file.size,
|
||||
mtimeMs,
|
||||
storage: { type: "tar", file },
|
||||
});
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
async function readZipEntries(source: ByteSource): Promise<ArchiveIndexEntry[]> {
|
||||
const directoryInfo = await readZipCentralDirectoryInfo(source);
|
||||
const centralDirectory = await source.read(directoryInfo.offset, directoryInfo.offset + directoryInfo.size);
|
||||
return parseZipCentralDirectory(source, centralDirectory, directoryInfo.entries);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split an `archive.ext:inner/path` reference into every plausible
|
||||
* `{ archivePath, subPath }` pair, longest archive prefix first. A path may
|
||||
* contain more than one archive extension, so each candidate is a guess at
|
||||
* where the archive ends and the member portion begins.
|
||||
*/
|
||||
export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] {
|
||||
const normalized = filePath.replace(/\\/g, "/");
|
||||
const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi;
|
||||
const seen = new Set<string>();
|
||||
const candidates: ArchivePathCandidate[] = [];
|
||||
|
||||
let match: RegExpExecArray | null;
|
||||
while (true) {
|
||||
match = pattern.exec(normalized);
|
||||
if (match === null) {
|
||||
break;
|
||||
}
|
||||
const end = match.index + match[0].length;
|
||||
const archivePath = filePath.slice(0, end);
|
||||
const subPath = normalized.slice(end).replace(/^:+/, "");
|
||||
const key = `${archivePath}\0${subPath}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
candidates.push({ archivePath, subPath });
|
||||
}
|
||||
|
||||
return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length);
|
||||
}
|
||||
|
||||
/**
|
||||
* An indexed, read-only view over a single archive. ZIP archives are indexed
|
||||
* from the central directory and members are inflated on demand; tar archives
|
||||
* are fully materialized by `Bun.Archive` up front.
|
||||
*/
|
||||
export class ArchiveReader {
|
||||
readonly format: ArchiveFormat;
|
||||
#entries = new Map<string, ArchiveIndexEntry>();
|
||||
|
||||
constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) {
|
||||
this.format = format;
|
||||
for (const entry of entries) {
|
||||
upsertArchiveEntry(this.#entries, entry);
|
||||
}
|
||||
ensureParentDirectories(this.#entries);
|
||||
}
|
||||
|
||||
getNode(subPath?: string): ArchiveNode | undefined {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (normalizedPath === undefined) return undefined;
|
||||
if (normalizedPath === "") {
|
||||
return { path: "", isDirectory: true, size: 0 };
|
||||
}
|
||||
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) return undefined;
|
||||
return {
|
||||
path: entry.path,
|
||||
isDirectory: entry.isDirectory,
|
||||
size: entry.size,
|
||||
mtimeMs: entry.mtimeMs,
|
||||
};
|
||||
}
|
||||
|
||||
listDirectory(subPath?: string): ArchiveDirectoryEntry[] {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (normalizedPath === undefined) {
|
||||
throw new ToolError("Archive path cannot contain '..'");
|
||||
}
|
||||
|
||||
if (normalizedPath) {
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' not found`);
|
||||
}
|
||||
if (!entry.isDirectory) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' is not a directory`);
|
||||
}
|
||||
}
|
||||
|
||||
const prefix = normalizedPath ? `${normalizedPath}/` : "";
|
||||
const children = new Map<string, ArchiveDirectoryEntry>();
|
||||
|
||||
for (const entry of this.#entries.values()) {
|
||||
if (normalizedPath) {
|
||||
if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue;
|
||||
}
|
||||
|
||||
const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path;
|
||||
const nextSegment = relativePath.split("/")[0];
|
||||
if (!nextSegment) continue;
|
||||
|
||||
const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment;
|
||||
if (children.has(childPath)) continue;
|
||||
|
||||
const childEntry = this.#entries.get(childPath);
|
||||
const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/");
|
||||
children.set(childPath, {
|
||||
name: nextSegment,
|
||||
path: childPath,
|
||||
isDirectory,
|
||||
size: isDirectory ? 0 : (childEntry?.size ?? entry.size),
|
||||
mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs,
|
||||
});
|
||||
}
|
||||
|
||||
return [...children.values()].sort((left, right) =>
|
||||
left.name.toLowerCase().localeCompare(right.name.toLowerCase()),
|
||||
);
|
||||
}
|
||||
|
||||
async readFile(subPath: string): Promise<ExtractedArchiveFile> {
|
||||
const normalizedPath = normalizeArchiveLookupPath(subPath);
|
||||
if (!normalizedPath) {
|
||||
throw new ToolError("Archive file path is required");
|
||||
}
|
||||
|
||||
const entry = this.#entries.get(normalizedPath);
|
||||
if (!entry) {
|
||||
throw new ToolError(`Archive file '${normalizedPath}' not found`);
|
||||
}
|
||||
if (entry.isDirectory) {
|
||||
throw new ToolError(`Archive path '${normalizedPath}' is a directory`);
|
||||
}
|
||||
if (!entry.storage) {
|
||||
throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`);
|
||||
}
|
||||
if (entry.size > MAX_ARCHIVE_MEMBER_BYTES) {
|
||||
throw new ToolError(
|
||||
`Archive member '${normalizedPath}' is too large to extract in memory (${formatBytes(entry.size)} > ${formatBytes(MAX_ARCHIVE_MEMBER_BYTES)} limit)`,
|
||||
);
|
||||
}
|
||||
|
||||
const bytes =
|
||||
entry.storage.type === "tar"
|
||||
? await entry.storage.file.bytes()
|
||||
: await readZipFileBytes(entry.storage, entry.size);
|
||||
|
||||
return {
|
||||
path: entry.path,
|
||||
isDirectory: false,
|
||||
size: entry.size,
|
||||
mtimeMs: entry.mtimeMs,
|
||||
bytes,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Open an archive for reading. ZIP archives opened from a path are indexed
|
||||
* lazily via ranged central-directory reads (members inflate on demand); tar
|
||||
* archives and in-memory ZIPs are read from a single buffer.
|
||||
*/
|
||||
export async function openArchive(source: ArchiveSource): Promise<ArchiveReader> {
|
||||
if (typeof source === "string") {
|
||||
const format = archiveFormatFromPath(source);
|
||||
if (!format) {
|
||||
throw new ToolError(`Unsupported archive format: ${source}`);
|
||||
}
|
||||
if (format === "zip") {
|
||||
return new ArchiveReader(format, await readZipEntries(fileByteSource(source)));
|
||||
}
|
||||
|
||||
const file = Bun.file(source);
|
||||
const archiveSize = file.size;
|
||||
if (archiveSize > MAX_TAR_ARCHIVE_BYTES) {
|
||||
throw new ToolError(
|
||||
`Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
|
||||
);
|
||||
}
|
||||
return new ArchiveReader(format, await readTarEntries(await file.bytes()));
|
||||
}
|
||||
|
||||
const { bytes, format } = source;
|
||||
if (format === "zip") {
|
||||
return new ArchiveReader(format, await readZipEntries(memoryByteSource(bytes)));
|
||||
}
|
||||
if (bytes.byteLength > MAX_TAR_ARCHIVE_BYTES) {
|
||||
throw new ToolError(
|
||||
`Archive is too large to read in memory (${formatBytes(bytes.byteLength)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
|
||||
);
|
||||
}
|
||||
return new ArchiveReader(format, await readTarEntries(bytes));
|
||||
}
|
||||
|
||||
/** Render the top-level entries of an in-memory archive as one line each. */
|
||||
export async function listArchiveRoot(
|
||||
bytes: Uint8Array,
|
||||
format: ArchiveFormat,
|
||||
opts: { limit?: number } = {},
|
||||
): Promise<string> {
|
||||
const archive = await openArchive({ bytes, format });
|
||||
const entries = archive.listDirectory("");
|
||||
const limitedEntries = opts.limit !== undefined && opts.limit > 0 ? entries.slice(0, opts.limit) : entries;
|
||||
const lines = formatArchiveEntryLines(limitedEntries);
|
||||
return lines.length > 0 ? lines.join("\n") : "(empty archive directory)";
|
||||
}
|
||||
|
||||
async function resolveArchiveBytes(source: ArchiveSource): Promise<{ bytes: Uint8Array; format: ArchiveFormat }> {
|
||||
if (typeof source !== "string") return source;
|
||||
const format = archiveFormatFromPath(source);
|
||||
if (!format) {
|
||||
throw new ToolError(`Unsupported archive format: ${source}`);
|
||||
}
|
||||
return { bytes: await Bun.file(source).bytes(), format };
|
||||
}
|
||||
|
||||
async function memberToBytes(content: ArchiveMemberContent): Promise<Uint8Array> {
|
||||
if (typeof content === "string") return ENCODER.encode(content);
|
||||
if (content instanceof Uint8Array) return content;
|
||||
return new Uint8Array(await content.arrayBuffer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Fully materialize every file member into a `path → content` map: ZIP members
|
||||
* are inflated via fflate, tar members are returned as lazy `File`s. Use this
|
||||
* when you need every entry (rewrite, extract); for browsing or single-member
|
||||
* reads prefer `openArchive`, which is lazy for ZIP.
|
||||
*/
|
||||
export async function readArchiveEntries(source: ArchiveSource): Promise<Map<string, ArchiveMemberContent>> {
|
||||
const { bytes, format } = await resolveArchiveBytes(source);
|
||||
const entries = new Map<string, ArchiveMemberContent>();
|
||||
if (format === "zip") {
|
||||
const unzipped = unzipSync(bytes);
|
||||
for (const name in unzipped) {
|
||||
entries.set(name.replace(/\\/g, "/"), unzipped[name]!);
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
const files = await new Bun.Archive(bytes).files();
|
||||
for (const [name, file] of files) {
|
||||
entries.set(name.replace(/\\/g, "/"), file);
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
|
||||
/**
|
||||
* Serialize `entries` into an archive of `format` and write it to `destPath`.
|
||||
* ZIP is built with fflate, tar / tar.gz with `Bun.Archive` (gzip for tar.gz).
|
||||
* String members are encoded as UTF-8.
|
||||
*/
|
||||
export async function writeArchive(
|
||||
destPath: string,
|
||||
format: ArchiveFormat,
|
||||
entries: Iterable<readonly [string, ArchiveMemberContent]>,
|
||||
): Promise<void> {
|
||||
if (format === "zip") {
|
||||
const record: Record<string, Uint8Array> = {};
|
||||
for (const [name, content] of entries) {
|
||||
record[name.replace(/\\/g, "/")] = await memberToBytes(content);
|
||||
}
|
||||
await Bun.write(destPath, zipSync(record));
|
||||
return;
|
||||
}
|
||||
|
||||
const record: Record<string, ArchiveMemberContent> = {};
|
||||
for (const [name, content] of entries) {
|
||||
record[name.replace(/\\/g, "/")] = content;
|
||||
}
|
||||
await Bun.Archive.write(destPath, record, format === "tar.gz" ? { compress: "gzip" } : undefined);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract every file member to `destDir`, creating parent directories as
|
||||
* needed. Entries that would escape `destDir` (via `..` or an absolute path)
|
||||
* are rejected. Returns the number of files written.
|
||||
*/
|
||||
export async function extractArchive(source: ArchiveSource, destDir: string): Promise<number> {
|
||||
const extractRoot = path.resolve(destDir);
|
||||
const entries = await readArchiveEntries(source);
|
||||
let count = 0;
|
||||
for (const [name, content] of entries) {
|
||||
if (name.endsWith("/")) continue;
|
||||
const outputPath = path.resolve(extractRoot, name);
|
||||
if (!outputPath.startsWith(extractRoot + path.sep)) {
|
||||
throw new ToolError(`Archive entry escapes extraction dir: ${name}`);
|
||||
}
|
||||
await Bun.write(outputPath, content);
|
||||
count++;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user