diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f66aebbc7..9e4fe1357 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1790,6 +1790,9 @@ - Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533)) - Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). - Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). +### Fixed + +- Fixed valid `.tar` and `.tar.gz` archive reads terminating omp through libarchive by parsing tar members in-process ([#4774](https://github.com/can1357/oh-my-pi/issues/4774)). ## [16.3.11] - 2026-07-06 diff --git a/packages/coding-agent/src/utils/zip.ts b/packages/coding-agent/src/utils/zip.ts index 947ad189f..f3290db45 100644 --- a/packages/coding-agent/src/utils/zip.ts +++ b/packages/coding-agent/src/utils/zip.ts @@ -1,10 +1,13 @@ // The single archive boundary for the codebase: ZIP (framed here, over the raw -// DEFLATE codec in `node:zlib`) and tar / tar.gz (via `Bun.Archive`). This is -// the ONLY module that frames ZIP containers or touches `Bun.Archive`; the +// DEFLATE codec in `node:zlib`) and tar / tar.gz (parsed in-process here, gzip +// via `node:zlib`; only archive *writing* uses `Bun.Archive`). This is the ONLY +// module that frames ZIP containers, parses tar, or touches `Bun.Archive`; the // markit document converters, the read/search/write tools, the URL fetcher, the // debug report bundler, and the tool-binary installer all go through here so // there is exactly one archive implementation to reason about. Do not parse or -// build ZIP/tar, or call `Bun.Archive`, anywhere else. +// build ZIP/tar, or call `Bun.Archive`, anywhere else. Tar *reads* deliberately +// avoid libarchive: its internal allocation-failure path aborts the whole +// process (#4774). import * as path from "node:path"; import * as zlib from "node:zlib"; import { formatBytes } from "@oh-my-pi/pi-utils"; @@ -44,9 +47,9 @@ export function unzip(bytes: Uint8Array): Unzipped { } /** - * Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into - * memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is - * exempt: it is read via ranged central-directory access. + * Cap on tar/tar.gz archives loaded fully into memory for in-process indexing + * (gzip input is bounded to this decompressed size). ZIP is exempt: it is read + * via ranged central-directory access. */ const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024; /** @@ -144,7 +147,14 @@ function memoryByteSource(buffer: Uint8Array): ByteSource { interface TarStorage { type: "tar"; - file: File; + buffer: Uint8Array; + dataOffset: number; + sparse: boolean; +} + +interface TarLinkStorage { + type: "tar-link"; + targetPath: string; } interface ZipStorage { @@ -156,7 +166,7 @@ interface ZipStorage { localHeaderOffset: number; } -type EntryStorage = TarStorage | ZipStorage; +type EntryStorage = TarStorage | TarLinkStorage | ZipStorage; interface ArchiveIndexEntry extends ArchiveNode { storage?: EntryStorage; @@ -604,38 +614,368 @@ async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): return decodeZipMember(compressedBytes, storage.compression, uncompressedSize); } -async function readTarEntries(bytes: Uint8Array): Promise { - let archive: Bun.Archive; - try { - archive = new Bun.Archive(bytes); - } catch (error) { - throw new ToolError(error instanceof Error ? error.message : String(error)); - } +const TAR_BLOCK_SIZE = 512; +const TAR_NAME_OFFSET = 0; +const TAR_NAME_LENGTH = 100; +const TAR_SIZE_OFFSET = 124; +const TAR_SIZE_LENGTH = 12; +const TAR_MTIME_OFFSET = 136; +const TAR_MTIME_LENGTH = 12; +const TAR_CHECKSUM_OFFSET = 148; +const TAR_CHECKSUM_LENGTH = 8; +const TAR_TYPEFLAG_OFFSET = 156; +const TAR_LINKNAME_OFFSET = 157; +const TAR_LINKNAME_LENGTH = 100; +const TAR_PREFIX_OFFSET = 345; +const TAR_PREFIX_LENGTH = 155; +const GZIP_MAGIC_0 = 0x1f; +const GZIP_MAGIC_1 = 0x8b; +const TAR_TEXT_DECODER = new TextDecoder(); - let files: Map; +/** + * Decompress a gzip stream in-process, bounded to the tar archive cap so a + * gzip bomb cannot inflate without limit. Non-gzip input passes through. + */ +function gunzipIfNeeded(bytes: Uint8Array): Uint8Array { + if (bytes.length >= 2 && bytes[0] === GZIP_MAGIC_0 && bytes[1] === GZIP_MAGIC_1) { + return new Uint8Array(zlib.gunzipSync(bytes, { maxOutputLength: MAX_TAR_ARCHIVE_BYTES })); + } + return bytes; +} + +/** Read a NUL-terminated tar header string, clamped to the buffer bounds. */ +function readTarString(buffer: Uint8Array, offset: number, length: number): string { + const limit = Math.min(offset + length, buffer.length); + let end = offset; + while (end < limit && buffer[end] !== 0) end++; + return TAR_TEXT_DECODER.decode(buffer.subarray(offset, end)); +} + +/** + * Read a tar numeric header field: GNU base-256 (high bit set) or the usual + * NUL/space-padded octal. Non-octal bytes are ignored so trailing padding does + * not corrupt the value. + */ +function readTarNumeric(buffer: Uint8Array, offset: number, length: number): number { + if (offset >= buffer.length) return 0; + if ((buffer[offset]! & 0x80) !== 0) { + let value = buffer[offset]! & 0x7f; + for (let i = 1; i < length; i++) value = value * 256 + (buffer[offset + i] ?? 0); + return value; + } + let value = 0; + for (let i = 0; i < length; i++) { + const c = buffer[offset + i]; + if (c !== undefined && c >= 0x30 && c <= 0x37) value = value * 8 + (c - 0x30); + } + return value; +} + +function isTarZeroBlock(buffer: Uint8Array, offset: number): boolean { + for (let i = 0; i < TAR_BLOCK_SIZE; i++) { + if (buffer[offset + i] !== 0) return false; + } + return true; +} + +/** Verify a tar header block's checksum (both unsigned and signed conventions). */ +function tarChecksumMatches(buffer: Uint8Array, offset: number): boolean { + const stored = readTarNumeric(buffer, offset + TAR_CHECKSUM_OFFSET, TAR_CHECKSUM_LENGTH); + let unsigned = 0; + let signed = 0; + for (let i = 0; i < TAR_BLOCK_SIZE; i++) { + const inChecksum = i >= TAR_CHECKSUM_OFFSET && i < TAR_CHECKSUM_OFFSET + TAR_CHECKSUM_LENGTH; + const byte = inChecksum ? 0x20 : (buffer[offset + i] ?? 0); + unsigned += byte; + signed += (byte << 24) >> 24; + } + return stored === unsigned || stored === signed; +} + +/** Parse a PAX extended-header payload into its `key → value` records. */ +function parsePaxRecords(data: Uint8Array): Map { + const attrs = new Map(); + let pos = 0; + while (pos < data.length) { + let space = pos; + while (space < data.length && data[space] !== 0x20) space++; + if (space >= data.length) break; + let length = 0; + let valid = false; + for (let i = pos; i < space; i++) { + const c = data[i]!; + if (c < 0x30 || c > 0x39) { + valid = false; + break; + } + length = length * 10 + (c - 0x30); + valid = true; + } + if (!valid || length <= 0 || pos + length > data.length) break; + const record = data.subarray(space + 1, pos + length - 1); + const eq = record.indexOf(0x3d); + if (eq >= 0) { + attrs.set(TAR_TEXT_DECODER.decode(record.subarray(0, eq)), TAR_TEXT_DECODER.decode(record.subarray(eq + 1))); + } + pos += length; + } + return attrs; +} + +function paxDeclaresSparse(pax: Map | undefined): boolean { + if (!pax) return false; + for (const key of pax.keys()) { + if (key.startsWith("GNU.sparse.")) return true; + } + return false; +} + +/** + * Index a tar (optionally gzip-compressed) archive entirely in TypeScript. + * Handles ustar/GNU/pax layouts, `./`-prefixed and `prefix`-split names, GNU + * `@LongLink` names/link targets, pax `path`/`linkpath`/`size` overrides, and + * hard links. This deliberately avoids `Bun.Archive`/libarchive, whose + * internal allocation-failure path calls `abort()` and takes down the whole + * process on a crafted or oversized member (#4774). Sparse members are + * indexed but flagged; reading their bytes throws a catchable `ToolError` + * rather than returning a misassembled payload. + */ +function readTarEntries(rawBytes: Uint8Array): ArchiveIndexEntry[] { + let buffer: Uint8Array; try { - files = await archive.files(); + buffer = gunzipIfNeeded(rawBytes); } catch (error) { throw new ToolError(error instanceof Error ? error.message : String(error)); } const entries: ArchiveIndexEntry[] = []; - for (const [rawPath, file] of files) { - const normalizedPath = normalizeArchiveEntryPath(rawPath); + let offset = 0; + let longName: string | undefined; + let longLink: string | undefined; + let pax: Map | undefined; + const pendingLinks = new Map(); + // A valid tar ends with a zero block. Track whether the fully buffered input + // reaches one so truncated archives never expose a partial index. + let sawTerminator = false; + + while (offset + TAR_BLOCK_SIZE <= buffer.length) { + if (isTarZeroBlock(buffer, offset)) { + sawTerminator = true; + break; + } + if (!tarChecksumMatches(buffer, offset)) { + throw new ToolError("Invalid or corrupt tar archive header"); + } + + const typeFlag = String.fromCharCode(buffer[offset + TAR_TYPEFLAG_OFFSET] || 0x30); + let size = readTarNumeric(buffer, offset + TAR_SIZE_OFFSET, TAR_SIZE_LENGTH); + let name = readTarString(buffer, offset + TAR_NAME_OFFSET, TAR_NAME_LENGTH); + const prefix = readTarString(buffer, offset + TAR_PREFIX_OFFSET, TAR_PREFIX_LENGTH); + if (prefix) name = `${prefix}/${name}`; + let linkName = readTarString(buffer, offset + TAR_LINKNAME_OFFSET, TAR_LINKNAME_LENGTH); + const mtime = readTarNumeric(buffer, offset + TAR_MTIME_OFFSET, TAR_MTIME_LENGTH); + + offset += TAR_BLOCK_SIZE; + const dataBlocks = Math.ceil(size / TAR_BLOCK_SIZE) * TAR_BLOCK_SIZE; + + // Metadata-only headers: consume their payload, remember it for the next + // file header, then continue. + if (typeFlag === "L") { + longName = readTarString(buffer, offset, size); + offset += dataBlocks; + continue; + } + if (typeFlag === "K") { + longLink = readTarString(buffer, offset, size); + offset += dataBlocks; + continue; + } + if (typeFlag === "x" || typeFlag === "X") { + pax = parsePaxRecords(buffer.subarray(offset, Math.min(offset + size, buffer.length))); + offset += dataBlocks; + continue; + } + if (typeFlag === "g") { + offset += dataBlocks; + continue; + } + + if (longName !== undefined) name = longName; + if (longLink !== undefined) linkName = longLink; + const paxPath = pax?.get("path"); + if (paxPath !== undefined) name = paxPath; + const paxLinkPath = pax?.get("linkpath"); + if (paxLinkPath !== undefined) linkName = paxLinkPath; + const paxSize = pax?.get("size"); + if (paxSize !== undefined) { + const parsed = Number.parseInt(paxSize, 10); + if (Number.isFinite(parsed) && parsed >= 0) size = parsed; + } + // GNU 1.0 sparse PAX stores the user-visible path in a dedicated record + // while the file header carries an internal `GNUSparseFile.NNN` name. + // Surface the real name so listings and `read :` resolve + // the member (its bytes are still rejected as sparse below). The header + // `size` remains the on-disk stored length that drives offset advance + // and truncation; `GNU.sparse.realsize` is display-only. + const paxSparseName = pax?.get("GNU.sparse.name"); + if (paxSparseName !== undefined) name = paxSparseName; + let displaySize = size; + const paxSparseRealSize = pax?.get("GNU.sparse.realsize"); + if (paxSparseRealSize !== undefined) { + const parsed = Number.parseInt(paxSparseRealSize, 10); + if (Number.isFinite(parsed) && parsed >= 0) displaySize = parsed; + } + const sparse = typeFlag === "S" || paxDeclaresSparse(pax); + const dataOffset = offset; + const memberDataBlocks = Math.ceil(size / TAR_BLOCK_SIZE) * TAR_BLOCK_SIZE; + offset += memberDataBlocks; + longName = undefined; + longLink = undefined; + pax = undefined; + + const isDirectory = typeFlag === "5" || name.endsWith("/"); + const normalizedPath = normalizeArchiveEntryPath(name); if (!normalizedPath) continue; - const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined; + const mtimeMs = mtime > 0 ? mtime * 1000 : undefined; + + if (isDirectory) { + entries.push({ path: normalizedPath, isDirectory: true, size: 0, mtimeMs }); + continue; + } + if (typeFlag === "1" || typeFlag === "2") { + const kind = typeFlag === "1" ? "hard link" : "symlink"; + const portableLinkName = linkName.replace(/\\/g, "/"); + // Symlinks resolve relative to their own directory; a target that + // stays inside the archive normalizes to a member path or "" (the + // archive root, e.g. `current -> .`). `undefined` means the target + // escapes the root (or is absolute) and is kept as a dangling link. + const targetPath = + typeFlag === "1" + ? normalizeArchiveEntryPath(portableLinkName) + : path.posix.isAbsolute(portableLinkName) + ? undefined + : normalizeArchiveLookupPath(path.posix.join(path.posix.dirname(normalizedPath), portableLinkName)); + const entry: ArchiveIndexEntry = { + path: normalizedPath, + isDirectory: false, + size: 0, + mtimeMs, + }; + if (targetPath === undefined) { + if (kind === "hard link") { + throw new ToolError(`Archive hard link '${normalizedPath}' has an invalid target`); + } + entry.storage = { type: "tar-link", targetPath: portableLinkName }; + entries.push(entry); + continue; + } + entries.push(entry); + pendingLinks.set(entry, { kind, targetPath }); + continue; + } + // Only regular-file typeflags carry inline data we can slice. + if (typeFlag !== "0" && typeFlag !== "\0" && typeFlag !== "7" && typeFlag !== "S") continue; + // A declared payload that runs past the buffer means the archive was + // truncated mid-member (e.g. a partial download). Reject it while + // indexing so a root listing does not present a truncated archive as a + // valid directory, with the failure only surfacing on a later member read. + if (dataOffset + memberDataBlocks > buffer.length) { + throw new ToolError(`Archive member '${normalizedPath}' is truncated`); + } entries.push({ path: normalizedPath, isDirectory: false, - size: file.size, + size: displaySize, mtimeMs, - storage: { type: "tar", file }, + storage: { type: "tar", buffer, dataOffset, sparse }, }); } + // Fully buffered tar reads must reach an end-of-archive zero block. Without + // one, even complete entries form only a partial listing: later members may + // have been cut off by a truncated download. For gzip-shaped non-tar input, + // this also gives fetch a catchable error so it can fall back to binary. + if (!sawTerminator) { + throw new ToolError("Not a valid tar archive: missing terminating zero block"); + } + + // Link records carry no data. Resolve file targets after all headers are + // indexed; directory symlinks remain one alias node and are traversed lazily + // by ArchiveReader so N files behind M aliases never inflate the index to + // N×M entries during a root listing. + if (pendingLinks.size > 0) { + const entriesByPath = new Map(); + for (const entry of entries) entriesByPath.set(entry.path, entry); + const unresolved = new Set(pendingLinks.keys()); + + while (unresolved.size > 0) { + let resolved = 0; + for (const entry of unresolved) { + const pending = pendingLinks.get(entry)!; + const target = entriesByPath.get(pending.targetPath); + if (target?.storage && !target.isDirectory) { + entry.size = target.size; + entry.storage = target.storage; + unresolved.delete(entry); + resolved++; + continue; + } + if (target && unresolved.has(target)) continue; + + // An empty target is the archive root, which is always a directory. + const targetPrefix = `${pending.targetPath}/`; + const targetIsDirectory = + pending.targetPath === "" || + target?.isDirectory === true || + entries.some(candidate => candidate.path.startsWith(targetPrefix)); + if (!targetIsDirectory) { + if (pending.kind === "symlink") { + entry.storage = { type: "tar-link", targetPath: pending.targetPath }; + unresolved.delete(entry); + resolved++; + continue; + } + const reason = target ? "unreadable member" : "missing member"; + throw new ToolError(`Archive hard link '${entry.path}' targets ${reason} '${pending.targetPath}'`); + } + if (pending.kind === "hard link") { + throw new ToolError(`Archive hard link '${entry.path}' targets directory '${pending.targetPath}'`); + } + + entry.isDirectory = true; + entry.storage = { type: "tar-link", targetPath: pending.targetPath }; + unresolved.delete(entry); + resolved++; + } + if (resolved === 0) { + throw new ToolError("Archive contains cyclic or unsupported links"); + } + } + } + return entries; } +/** + * Slice one indexed tar member's bytes out of the archive buffer. Sparse + * members cannot be reassembled from a contiguous slice, so reading them throws + * a catchable error instead of returning corrupt data. + */ +function extractTarMember(storage: TarStorage, size: number, memberPath: string): Uint8Array { + if (storage.sparse) { + throw new ToolError(`Archive member '${memberPath}' is a sparse file and cannot be read`); + } + const end = storage.dataOffset + size; + if (end > storage.buffer.length) { + throw new ToolError(`Archive member '${memberPath}' is truncated`); + } + return storage.buffer.subarray(storage.dataOffset, end); +} + +function throwUnreadableTarLink(storage: TarLinkStorage, memberPath: string): never { + throw new ToolError(`Archive symlink '${memberPath}' cannot be materialized from target '${storage.targetPath}'`); +} + async function readZipEntries(source: ByteSource): Promise { const directoryInfo = await readZipCentralDirectoryInfo(source); const centralDirectory = await source.read(directoryInfo.offset, directoryInfo.offset + directoryInfo.size); @@ -675,7 +1015,8 @@ export function parseArchivePathCandidates(filePath: string): ArchivePathCandida /** * An indexed, read-only view over a single archive. ZIP archives are indexed * from the central directory and members are inflated on demand; tar archives - * are fully materialized by `Bun.Archive` up front. + * are parsed from one in-memory buffer, members are sliced on demand, and + * directory symlink aliases are traversed lazily. */ export class ArchiveReader { readonly format: ArchiveFormat; @@ -689,6 +1030,34 @@ export class ArchiveReader { ensureParentDirectories(this.#entries); } + #resolveDirectoryAliases(archivePath: string): string { + let resolvedPath = archivePath; + const seen = new Set(); + while (true) { + if (seen.has(resolvedPath)) { + throw new ToolError(`Archive path '${archivePath}' crosses a cyclic symlink`); + } + seen.add(resolvedPath); + + const parts = resolvedPath.split("/"); + let replacement: string | undefined; + for (let end = parts.length; end > 0; end--) { + const prefix = parts.slice(0, end).join("/"); + const entry = this.#entries.get(prefix); + if (!entry?.isDirectory || entry.storage?.type !== "tar-link") continue; + const suffix = parts.slice(end).join("/"); + replacement = suffix + ? entry.storage.targetPath + ? `${entry.storage.targetPath}/${suffix}` + : suffix + : entry.storage.targetPath; + break; + } + if (replacement === undefined) return resolvedPath; + resolvedPath = replacement; + } + } + getNode(subPath?: string): ArchiveNode | undefined { const normalizedPath = normalizeArchiveLookupPath(subPath); if (normalizedPath === undefined) return undefined; @@ -696,10 +1065,14 @@ export class ArchiveReader { return { path: "", isDirectory: true, size: 0 }; } - const entry = this.#entries.get(normalizedPath); + const resolvedPath = this.#resolveDirectoryAliases(normalizedPath); + if (resolvedPath === "") { + return { path: normalizedPath, isDirectory: true, size: 0 }; + } + const entry = this.#entries.get(resolvedPath); if (!entry) return undefined; return { - path: entry.path, + path: normalizedPath, isDirectory: entry.isDirectory, size: entry.size, mtimeMs: entry.mtimeMs, @@ -712,8 +1085,9 @@ export class ArchiveReader { throw new ToolError("Archive path cannot contain '..'"); } - if (normalizedPath) { - const entry = this.#entries.get(normalizedPath); + const resolvedPath = normalizedPath ? this.#resolveDirectoryAliases(normalizedPath) : ""; + if (normalizedPath && resolvedPath !== "") { + const entry = this.#entries.get(resolvedPath); if (!entry) { throw new ToolError(`Archive path '${normalizedPath}' not found`); } @@ -722,22 +1096,23 @@ export class ArchiveReader { } } - const prefix = normalizedPath ? `${normalizedPath}/` : ""; + const sourcePrefix = resolvedPath ? `${resolvedPath}/` : ""; const children = new Map(); for (const entry of this.#entries.values()) { - if (normalizedPath) { - if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue; + if (resolvedPath) { + if (!entry.path.startsWith(sourcePrefix) || entry.path === resolvedPath) continue; } - const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path; + const relativePath = resolvedPath ? entry.path.slice(sourcePrefix.length) : entry.path; const nextSegment = relativePath.split("/")[0]; if (!nextSegment) continue; const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment; if (children.has(childPath)) continue; - const childEntry = this.#entries.get(childPath); + const sourceChildPath = resolvedPath ? `${resolvedPath}/${nextSegment}` : nextSegment; + const childEntry = this.#entries.get(sourceChildPath); const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/"); children.set(childPath, { name: nextSegment, @@ -759,7 +1134,11 @@ export class ArchiveReader { throw new ToolError("Archive file path is required"); } - const entry = this.#entries.get(normalizedPath); + const resolvedPath = this.#resolveDirectoryAliases(normalizedPath); + if (resolvedPath === "") { + throw new ToolError(`Archive path '${normalizedPath}' is a directory`); + } + const entry = this.#entries.get(resolvedPath); if (!entry) { throw new ToolError(`Archive file '${normalizedPath}' not found`); } @@ -775,13 +1154,16 @@ export class ArchiveReader { ); } - const bytes = - entry.storage.type === "tar" - ? await entry.storage.file.bytes() - : await readZipFileBytes(entry.storage, entry.size); - + let bytes: Uint8Array; + if (entry.storage.type === "tar") { + bytes = extractTarMember(entry.storage, entry.size, normalizedPath); + } else if (entry.storage.type === "tar-link") { + throwUnreadableTarLink(entry.storage, normalizedPath); + } else { + bytes = await readZipFileBytes(entry.storage, entry.size); + } return { - path: entry.path, + path: normalizedPath, isDirectory: false, size: entry.size, mtimeMs: entry.mtimeMs, @@ -812,7 +1194,7 @@ export async function openArchive(source: ArchiveSource): Promise `Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`, ); } - return new ArchiveReader(format, await readTarEntries(await file.bytes())); + return new ArchiveReader(format, readTarEntries(await file.bytes())); } const { bytes, format } = source; @@ -824,7 +1206,7 @@ export async function openArchive(source: ArchiveSource): Promise `Archive is too large to read in memory (${formatBytes(bytes.byteLength)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`, ); } - return new ArchiveReader(format, await readTarEntries(bytes)); + return new ArchiveReader(format, readTarEntries(bytes)); } /** Render the top-level entries of an in-memory archive as one line each. */ @@ -857,9 +1239,9 @@ async function memberToBytes(content: ArchiveMemberContent): Promise /** * Fully materialize every file member into a `path → content` map: ZIP members - * are inflated in memory, tar members are returned as lazy `File`s. Use this - * when you need every entry (rewrite, extract); for browsing or single-member - * reads prefer `openArchive`, which is lazy for ZIP. + * are inflated in memory, tar members are sliced from the decoded archive + * buffer. Use this when you need every entry (rewrite, extract); for browsing + * or single-member reads prefer `openArchive`, which is lazy for ZIP. */ export async function readArchiveEntries(source: ArchiveSource): Promise> { const { bytes, format } = await resolveArchiveBytes(source); @@ -871,9 +1253,23 @@ export async function readArchiveEntries(source: ArchiveSource): Promise { const table = new Uint32Array(256); for (let index = 0; index < 256; index++) { @@ -753,6 +819,203 @@ describe("Coding Agent Tools", () => { expect(result.details?.isDirectory).toBe(true); }); + it("should read tar.gz members with UTF-8 ustar prefixes", async () => { + const archivePath = path.join(testDir, "unicode-prefix.tar.gz"); + const prefix = "bun-da3851e57ae130c5594d0e208a5da5ba8c13edfb/test/js/node/test/fixtures/copy/utf/新建文件夹"; + const memberPath = `${prefix}/experimental.json`; + fs.writeFileSync( + archivePath, + zlib.gzipSync( + createTarArchive([ + { + path: "experimental.json", + prefix, + content: '{ "type": "module" }', + }, + ]), + ), + ); + + const rootResult = await readTool.execute("test-call-tar-unicode-prefix-root", { path: archivePath }); + expect(getTextOutput(rootResult)).toContain("bun-da3851e57ae130c5594d0e208a5da5ba8c13edfb/"); + expect(rootResult.details?.isDirectory).toBe(true); + + const memberResult = await readTool.execute("test-call-tar-unicode-prefix-member", { + path: `${archivePath}:${memberPath}`, + }); + expect(getTextOutput(memberResult)).toContain('{ "type": "module" }'); + }); + + it("should preserve tar hard-link members", async () => { + const archivePath = path.join(testDir, "hard-link.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([ + { path: "pkg/original.txt", content: "shared content\n" }, + { path: "pkg/linked.txt", content: "", typeFlag: "1", linkName: "pkg/original.txt" }, + ]), + ); + + const rootResult = await readTool.execute("test-call-tar-hard-link-root", { path: `${archivePath}:pkg` }); + expect(getTextOutput(rootResult)).toContain("linked.txt"); + + const linkedResult = await readTool.execute("test-call-tar-hard-link-member", { + path: `${archivePath}:pkg/linked.txt`, + }); + expect(getTextOutput(linkedResult)).toContain("shared content"); + + const entries = await readArchiveEntries(archivePath); + const linkedContent = entries.get("pkg/linked.txt"); + if (!(linkedContent instanceof Uint8Array)) { + throw new Error("Expected hard-link content to materialize as bytes"); + } + expect(new TextDecoder().decode(linkedContent)).toBe("shared content\n"); + }); + + it("should preserve safe relative tar file symlinks", async () => { + const archivePath = path.join(testDir, "file-symlink.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([ + { path: "pkg/lib/tool.js", content: "export const linked = true;\n" }, + { path: "pkg/bin/tool", content: "", typeFlag: "2", linkName: "../lib/tool.js" }, + ]), + ); + + const linkedResult = await readTool.execute("test-call-tar-symlink-member", { + path: `${archivePath}:pkg/bin/tool`, + }); + expect(getTextOutput(linkedResult)).toContain("export const linked = true"); + + const entries = await readArchiveEntries(archivePath); + const linkedContent = entries.get("pkg/bin/tool"); + if (!(linkedContent instanceof Uint8Array)) { + throw new Error("Expected symlink content to materialize as bytes"); + } + expect(new TextDecoder().decode(linkedContent)).toBe("export const linked = true;\n"); + }); + + it("should resolve directory symlinks lazily without materializing subtrees", async () => { + const archivePath = path.join(testDir, "directory-symlinks.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([ + { path: "pkg/lib/tool.js", content: "export const linked = true;\n" }, + { path: "pkg/lib/extra.js", content: "export const extra = true;\n" }, + { path: "pkg/current-a", content: "", typeFlag: "2", linkName: "lib" }, + { path: "pkg/current-b", content: "", typeFlag: "2", linkName: "lib" }, + { path: "pkg/current-c", content: "", typeFlag: "2", linkName: "lib" }, + ]), + ); + + const linkedResult = await readTool.execute("test-call-tar-directory-symlink-member", { + path: `${archivePath}:pkg/current-a/tool.js`, + }); + expect(getTextOutput(linkedResult)).toContain("export const linked = true"); + + const directoryResult = await readTool.execute("test-call-tar-directory-symlink-directory", { + path: `${archivePath}:pkg/current-b`, + }); + expect(getTextOutput(directoryResult)).toContain("extra.js"); + + await expect(readArchiveEntries(archivePath)).rejects.toThrow(/cannot be materialized/); + }); + + it("should resolve tar symlinks whose target is the archive root", async () => { + const archivePath = path.join(testDir, "root-symlinks.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([ + { path: "top.txt", content: "top level\n" }, + { path: "dir/inner.txt", content: "inner\n" }, + // `current -> .` and `dir/up -> ..` both normalize to the archive root. + { path: "current", content: "", typeFlag: "2", linkName: "." }, + { path: "dir/up", content: "", typeFlag: "2", linkName: ".." }, + ]), + ); + + const currentNode = await readTool.execute("test-call-tar-root-symlink-current", { + path: `${archivePath}:current/top.txt`, + }); + expect(getTextOutput(currentNode)).toContain("top level"); + + const upNode = await readTool.execute("test-call-tar-root-symlink-up", { + path: `${archivePath}:dir/up/top.txt`, + }); + expect(getTextOutput(upNode)).toContain("top level"); + }); + + it("should list dangling tar symlinks but reject their materialization", async () => { + const archivePath = path.join(testDir, "dangling-symlink.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([{ path: "pkg/dangling", content: "", typeFlag: "2", linkName: "missing-target" }]), + ); + + const rootResult = await readTool.execute("test-call-tar-dangling-symlink-root", { + path: `${archivePath}:pkg`, + }); + expect(getTextOutput(rootResult)).toContain("dangling"); + await expect( + readTool.execute("test-call-tar-dangling-symlink-member", { + path: `${archivePath}:pkg/dangling`, + }), + ).rejects.toThrow(/cannot be materialized/); + await expect(readArchiveEntries(archivePath)).rejects.toThrow(/cannot be materialized/); + }); + + it("should surface GNU sparse PAX names and reject sparse reads", async () => { + const archivePath = path.join(testDir, "sparse-pax.tar"); + fs.writeFileSync( + archivePath, + createSparsePaxTarArchive("data/sparse.bin", 1048576, Buffer.from("sparse-map\n")), + ); + + // The listing must show the real GNU.sparse.name, not the internal + // GNUSparseFile.NNN path. + const rootResult = await readTool.execute("test-call-tar-sparse-root", { path: `${archivePath}:data` }); + expect(getTextOutput(rootResult)).toContain("sparse.bin"); + expect(getTextOutput(rootResult)).not.toContain("GNUSparseFile"); + + // Reading the real member name resolves the entry and rejects it as + // sparse (a catchable error), rather than reporting it missing. + await expect( + readTool.execute("test-call-tar-sparse-member", { path: `${archivePath}:data/sparse.bin` }), + ).rejects.toThrow(/sparse file and cannot be read/); + }); + + it("should reject a truncated tar member while indexing", async () => { + const archivePath = path.join(testDir, "truncated.tar"); + // A full, valid archive declares 2048 bytes for `big.txt`; slicing the + // payload mid-member leaves the header's declared size pointing past EOF. + const complete = createTarArchive([{ path: "big.txt", content: "A".repeat(2048) }]); + fs.writeFileSync(archivePath, complete.subarray(0, 512 + 256)); + + await expect(readTool.execute("test-call-tar-truncated", { path: archivePath })).rejects.toThrow(/truncated/); + }); + + it("should reject a tar truncated before its terminating zero block", async () => { + const archivePath = path.join(testDir, "unterminated.tar"); + const complete = createTarArchive([{ path: "complete.txt", content: "complete member\n" }]); + fs.writeFileSync(archivePath, complete.subarray(0, complete.length - 1024)); + + await expect(readTool.execute("test-call-tar-unterminated", { path: archivePath })).rejects.toThrow( + /missing terminating zero block/, + ); + }); + + it("should reject a gzip payload that is not a tar archive", async () => { + // `sniffArchiveFormat` classifies any gzip magic as tar.gz, so a plain + // `.txt.gz` (decompressed payload shorter than one 512-byte tar block) + // must raise a catchable error instead of listing an empty directory. + const archivePath = path.join(testDir, "note.tar.gz"); + fs.writeFileSync(archivePath, zlib.gzipSync(Buffer.from("hello world\n"))); + + await expect(readTool.execute("test-call-gzip-non-tar", { path: archivePath })).rejects.toThrow( + /not a valid tar archive/i, + ); + }); + it("should list archive subdirectories", async () => { const archivePath = path.join(testDir, "fixture.zip"); fs.writeFileSync(