Merge PR #8260: fix(read): avoid libarchive for tar reads (@roboomp)

This commit is contained in:
can1357
2026-08-12 01:53:49 +02:00
3 changed files with 711 additions and 49 deletions
+3
View File
@@ -1790,6 +1790,9 @@
- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533))
- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)).
- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)).
### Fixed
- Fixed valid `.tar` and `.tar.gz` archive reads terminating omp through libarchive by parsing tar members in-process ([#4774](https://github.com/can1357/oh-my-pi/issues/4774)).
## [16.3.11] - 2026-07-06
+443 -47
View File
@@ -1,10 +1,13 @@
// The single archive boundary for the codebase: ZIP (framed here, over the raw
// DEFLATE codec in `node:zlib`) and tar / tar.gz (via `Bun.Archive`). This is
// the ONLY module that frames ZIP containers or touches `Bun.Archive`; the
// DEFLATE codec in `node:zlib`) and tar / tar.gz (parsed in-process here, gzip
// via `node:zlib`; only archive *writing* uses `Bun.Archive`). This is the ONLY
// module that frames ZIP containers, parses tar, or touches `Bun.Archive`; the
// markit document converters, the read/search/write tools, the URL fetcher, the
// debug report bundler, and the tool-binary installer all go through here so
// there is exactly one archive implementation to reason about. Do not parse or
// build ZIP/tar, or call `Bun.Archive`, anywhere else.
// build ZIP/tar, or call `Bun.Archive`, anywhere else. Tar *reads* deliberately
// avoid libarchive: its internal allocation-failure path aborts the whole
// process (#4774).
import * as path from "node:path";
import * as zlib from "node:zlib";
import { formatBytes } from "@oh-my-pi/pi-utils";
@@ -44,9 +47,9 @@ export function unzip(bytes: Uint8Array): Unzipped {
}
/**
* Cap on the on-disk size of tar/tar.gz archives, which are loaded fully into
* memory (and decompressed by `Bun.Archive`) just to index entries. ZIP is
* exempt: it is read via ranged central-directory access.
* Cap on tar/tar.gz archives loaded fully into memory for in-process indexing
* (gzip input is bounded to this decompressed size). ZIP is exempt: it is read
* via ranged central-directory access.
*/
const MAX_TAR_ARCHIVE_BYTES = 256 * 1024 * 1024;
/**
@@ -144,7 +147,14 @@ function memoryByteSource(buffer: Uint8Array): ByteSource {
interface TarStorage {
type: "tar";
file: File;
buffer: Uint8Array;
dataOffset: number;
sparse: boolean;
}
interface TarLinkStorage {
type: "tar-link";
targetPath: string;
}
interface ZipStorage {
@@ -156,7 +166,7 @@ interface ZipStorage {
localHeaderOffset: number;
}
type EntryStorage = TarStorage | ZipStorage;
type EntryStorage = TarStorage | TarLinkStorage | ZipStorage;
interface ArchiveIndexEntry extends ArchiveNode {
storage?: EntryStorage;
@@ -604,38 +614,368 @@ async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number):
return decodeZipMember(compressedBytes, storage.compression, uncompressedSize);
}
async function readTarEntries(bytes: Uint8Array): Promise<ArchiveIndexEntry[]> {
let archive: Bun.Archive;
try {
archive = new Bun.Archive(bytes);
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
const TAR_BLOCK_SIZE = 512;
const TAR_NAME_OFFSET = 0;
const TAR_NAME_LENGTH = 100;
const TAR_SIZE_OFFSET = 124;
const TAR_SIZE_LENGTH = 12;
const TAR_MTIME_OFFSET = 136;
const TAR_MTIME_LENGTH = 12;
const TAR_CHECKSUM_OFFSET = 148;
const TAR_CHECKSUM_LENGTH = 8;
const TAR_TYPEFLAG_OFFSET = 156;
const TAR_LINKNAME_OFFSET = 157;
const TAR_LINKNAME_LENGTH = 100;
const TAR_PREFIX_OFFSET = 345;
const TAR_PREFIX_LENGTH = 155;
const GZIP_MAGIC_0 = 0x1f;
const GZIP_MAGIC_1 = 0x8b;
const TAR_TEXT_DECODER = new TextDecoder();
let files: Map<string, File>;
/**
* Decompress a gzip stream in-process, bounded to the tar archive cap so a
* gzip bomb cannot inflate without limit. Non-gzip input passes through.
*/
function gunzipIfNeeded(bytes: Uint8Array): Uint8Array {
if (bytes.length >= 2 && bytes[0] === GZIP_MAGIC_0 && bytes[1] === GZIP_MAGIC_1) {
return new Uint8Array(zlib.gunzipSync(bytes, { maxOutputLength: MAX_TAR_ARCHIVE_BYTES }));
}
return bytes;
}
/** Read a NUL-terminated tar header string, clamped to the buffer bounds. */
function readTarString(buffer: Uint8Array, offset: number, length: number): string {
const limit = Math.min(offset + length, buffer.length);
let end = offset;
while (end < limit && buffer[end] !== 0) end++;
return TAR_TEXT_DECODER.decode(buffer.subarray(offset, end));
}
/**
* Read a tar numeric header field: GNU base-256 (high bit set) or the usual
* NUL/space-padded octal. Non-octal bytes are ignored so trailing padding does
* not corrupt the value.
*/
function readTarNumeric(buffer: Uint8Array, offset: number, length: number): number {
if (offset >= buffer.length) return 0;
if ((buffer[offset]! & 0x80) !== 0) {
let value = buffer[offset]! & 0x7f;
for (let i = 1; i < length; i++) value = value * 256 + (buffer[offset + i] ?? 0);
return value;
}
let value = 0;
for (let i = 0; i < length; i++) {
const c = buffer[offset + i];
if (c !== undefined && c >= 0x30 && c <= 0x37) value = value * 8 + (c - 0x30);
}
return value;
}
function isTarZeroBlock(buffer: Uint8Array, offset: number): boolean {
for (let i = 0; i < TAR_BLOCK_SIZE; i++) {
if (buffer[offset + i] !== 0) return false;
}
return true;
}
/** Verify a tar header block's checksum (both unsigned and signed conventions). */
function tarChecksumMatches(buffer: Uint8Array, offset: number): boolean {
const stored = readTarNumeric(buffer, offset + TAR_CHECKSUM_OFFSET, TAR_CHECKSUM_LENGTH);
let unsigned = 0;
let signed = 0;
for (let i = 0; i < TAR_BLOCK_SIZE; i++) {
const inChecksum = i >= TAR_CHECKSUM_OFFSET && i < TAR_CHECKSUM_OFFSET + TAR_CHECKSUM_LENGTH;
const byte = inChecksum ? 0x20 : (buffer[offset + i] ?? 0);
unsigned += byte;
signed += (byte << 24) >> 24;
}
return stored === unsigned || stored === signed;
}
/** Parse a PAX extended-header payload into its `key → value` records. */
function parsePaxRecords(data: Uint8Array): Map<string, string> {
const attrs = new Map<string, string>();
let pos = 0;
while (pos < data.length) {
let space = pos;
while (space < data.length && data[space] !== 0x20) space++;
if (space >= data.length) break;
let length = 0;
let valid = false;
for (let i = pos; i < space; i++) {
const c = data[i]!;
if (c < 0x30 || c > 0x39) {
valid = false;
break;
}
length = length * 10 + (c - 0x30);
valid = true;
}
if (!valid || length <= 0 || pos + length > data.length) break;
const record = data.subarray(space + 1, pos + length - 1);
const eq = record.indexOf(0x3d);
if (eq >= 0) {
attrs.set(TAR_TEXT_DECODER.decode(record.subarray(0, eq)), TAR_TEXT_DECODER.decode(record.subarray(eq + 1)));
}
pos += length;
}
return attrs;
}
function paxDeclaresSparse(pax: Map<string, string> | undefined): boolean {
if (!pax) return false;
for (const key of pax.keys()) {
if (key.startsWith("GNU.sparse.")) return true;
}
return false;
}
/**
* Index a tar (optionally gzip-compressed) archive entirely in TypeScript.
* Handles ustar/GNU/pax layouts, `./`-prefixed and `prefix`-split names, GNU
* `@LongLink` names/link targets, pax `path`/`linkpath`/`size` overrides, and
* hard links. This deliberately avoids `Bun.Archive`/libarchive, whose
* internal allocation-failure path calls `abort()` and takes down the whole
* process on a crafted or oversized member (#4774). Sparse members are
* indexed but flagged; reading their bytes throws a catchable `ToolError`
* rather than returning a misassembled payload.
*/
function readTarEntries(rawBytes: Uint8Array): ArchiveIndexEntry[] {
let buffer: Uint8Array;
try {
files = await archive.files();
buffer = gunzipIfNeeded(rawBytes);
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
const entries: ArchiveIndexEntry[] = [];
for (const [rawPath, file] of files) {
const normalizedPath = normalizeArchiveEntryPath(rawPath);
let offset = 0;
let longName: string | undefined;
let longLink: string | undefined;
let pax: Map<string, string> | undefined;
const pendingLinks = new Map<ArchiveIndexEntry, { kind: "hard link" | "symlink"; targetPath: string }>();
// A valid tar ends with a zero block. Track whether the fully buffered input
// reaches one so truncated archives never expose a partial index.
let sawTerminator = false;
while (offset + TAR_BLOCK_SIZE <= buffer.length) {
if (isTarZeroBlock(buffer, offset)) {
sawTerminator = true;
break;
}
if (!tarChecksumMatches(buffer, offset)) {
throw new ToolError("Invalid or corrupt tar archive header");
}
const typeFlag = String.fromCharCode(buffer[offset + TAR_TYPEFLAG_OFFSET] || 0x30);
let size = readTarNumeric(buffer, offset + TAR_SIZE_OFFSET, TAR_SIZE_LENGTH);
let name = readTarString(buffer, offset + TAR_NAME_OFFSET, TAR_NAME_LENGTH);
const prefix = readTarString(buffer, offset + TAR_PREFIX_OFFSET, TAR_PREFIX_LENGTH);
if (prefix) name = `${prefix}/${name}`;
let linkName = readTarString(buffer, offset + TAR_LINKNAME_OFFSET, TAR_LINKNAME_LENGTH);
const mtime = readTarNumeric(buffer, offset + TAR_MTIME_OFFSET, TAR_MTIME_LENGTH);
offset += TAR_BLOCK_SIZE;
const dataBlocks = Math.ceil(size / TAR_BLOCK_SIZE) * TAR_BLOCK_SIZE;
// Metadata-only headers: consume their payload, remember it for the next
// file header, then continue.
if (typeFlag === "L") {
longName = readTarString(buffer, offset, size);
offset += dataBlocks;
continue;
}
if (typeFlag === "K") {
longLink = readTarString(buffer, offset, size);
offset += dataBlocks;
continue;
}
if (typeFlag === "x" || typeFlag === "X") {
pax = parsePaxRecords(buffer.subarray(offset, Math.min(offset + size, buffer.length)));
offset += dataBlocks;
continue;
}
if (typeFlag === "g") {
offset += dataBlocks;
continue;
}
if (longName !== undefined) name = longName;
if (longLink !== undefined) linkName = longLink;
const paxPath = pax?.get("path");
if (paxPath !== undefined) name = paxPath;
const paxLinkPath = pax?.get("linkpath");
if (paxLinkPath !== undefined) linkName = paxLinkPath;
const paxSize = pax?.get("size");
if (paxSize !== undefined) {
const parsed = Number.parseInt(paxSize, 10);
if (Number.isFinite(parsed) && parsed >= 0) size = parsed;
}
// GNU 1.0 sparse PAX stores the user-visible path in a dedicated record
// while the file header carries an internal `GNUSparseFile.NNN` name.
// Surface the real name so listings and `read <archive>:<name>` resolve
// the member (its bytes are still rejected as sparse below). The header
// `size` remains the on-disk stored length that drives offset advance
// and truncation; `GNU.sparse.realsize` is display-only.
const paxSparseName = pax?.get("GNU.sparse.name");
if (paxSparseName !== undefined) name = paxSparseName;
let displaySize = size;
const paxSparseRealSize = pax?.get("GNU.sparse.realsize");
if (paxSparseRealSize !== undefined) {
const parsed = Number.parseInt(paxSparseRealSize, 10);
if (Number.isFinite(parsed) && parsed >= 0) displaySize = parsed;
}
const sparse = typeFlag === "S" || paxDeclaresSparse(pax);
const dataOffset = offset;
const memberDataBlocks = Math.ceil(size / TAR_BLOCK_SIZE) * TAR_BLOCK_SIZE;
offset += memberDataBlocks;
longName = undefined;
longLink = undefined;
pax = undefined;
const isDirectory = typeFlag === "5" || name.endsWith("/");
const normalizedPath = normalizeArchiveEntryPath(name);
if (!normalizedPath) continue;
const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined;
const mtimeMs = mtime > 0 ? mtime * 1000 : undefined;
if (isDirectory) {
entries.push({ path: normalizedPath, isDirectory: true, size: 0, mtimeMs });
continue;
}
if (typeFlag === "1" || typeFlag === "2") {
const kind = typeFlag === "1" ? "hard link" : "symlink";
const portableLinkName = linkName.replace(/\\/g, "/");
// Symlinks resolve relative to their own directory; a target that
// stays inside the archive normalizes to a member path or "" (the
// archive root, e.g. `current -> .`). `undefined` means the target
// escapes the root (or is absolute) and is kept as a dangling link.
const targetPath =
typeFlag === "1"
? normalizeArchiveEntryPath(portableLinkName)
: path.posix.isAbsolute(portableLinkName)
? undefined
: normalizeArchiveLookupPath(path.posix.join(path.posix.dirname(normalizedPath), portableLinkName));
const entry: ArchiveIndexEntry = {
path: normalizedPath,
isDirectory: false,
size: 0,
mtimeMs,
};
if (targetPath === undefined) {
if (kind === "hard link") {
throw new ToolError(`Archive hard link '${normalizedPath}' has an invalid target`);
}
entry.storage = { type: "tar-link", targetPath: portableLinkName };
entries.push(entry);
continue;
}
entries.push(entry);
pendingLinks.set(entry, { kind, targetPath });
continue;
}
// Only regular-file typeflags carry inline data we can slice.
if (typeFlag !== "0" && typeFlag !== "\0" && typeFlag !== "7" && typeFlag !== "S") continue;
// A declared payload that runs past the buffer means the archive was
// truncated mid-member (e.g. a partial download). Reject it while
// indexing so a root listing does not present a truncated archive as a
// valid directory, with the failure only surfacing on a later member read.
if (dataOffset + memberDataBlocks > buffer.length) {
throw new ToolError(`Archive member '${normalizedPath}' is truncated`);
}
entries.push({
path: normalizedPath,
isDirectory: false,
size: file.size,
size: displaySize,
mtimeMs,
storage: { type: "tar", file },
storage: { type: "tar", buffer, dataOffset, sparse },
});
}
// Fully buffered tar reads must reach an end-of-archive zero block. Without
// one, even complete entries form only a partial listing: later members may
// have been cut off by a truncated download. For gzip-shaped non-tar input,
// this also gives fetch a catchable error so it can fall back to binary.
if (!sawTerminator) {
throw new ToolError("Not a valid tar archive: missing terminating zero block");
}
// Link records carry no data. Resolve file targets after all headers are
// indexed; directory symlinks remain one alias node and are traversed lazily
// by ArchiveReader so N files behind M aliases never inflate the index to
// N×M entries during a root listing.
if (pendingLinks.size > 0) {
const entriesByPath = new Map<string, ArchiveIndexEntry>();
for (const entry of entries) entriesByPath.set(entry.path, entry);
const unresolved = new Set(pendingLinks.keys());
while (unresolved.size > 0) {
let resolved = 0;
for (const entry of unresolved) {
const pending = pendingLinks.get(entry)!;
const target = entriesByPath.get(pending.targetPath);
if (target?.storage && !target.isDirectory) {
entry.size = target.size;
entry.storage = target.storage;
unresolved.delete(entry);
resolved++;
continue;
}
if (target && unresolved.has(target)) continue;
// An empty target is the archive root, which is always a directory.
const targetPrefix = `${pending.targetPath}/`;
const targetIsDirectory =
pending.targetPath === "" ||
target?.isDirectory === true ||
entries.some(candidate => candidate.path.startsWith(targetPrefix));
if (!targetIsDirectory) {
if (pending.kind === "symlink") {
entry.storage = { type: "tar-link", targetPath: pending.targetPath };
unresolved.delete(entry);
resolved++;
continue;
}
const reason = target ? "unreadable member" : "missing member";
throw new ToolError(`Archive hard link '${entry.path}' targets ${reason} '${pending.targetPath}'`);
}
if (pending.kind === "hard link") {
throw new ToolError(`Archive hard link '${entry.path}' targets directory '${pending.targetPath}'`);
}
entry.isDirectory = true;
entry.storage = { type: "tar-link", targetPath: pending.targetPath };
unresolved.delete(entry);
resolved++;
}
if (resolved === 0) {
throw new ToolError("Archive contains cyclic or unsupported links");
}
}
}
return entries;
}
/**
* Slice one indexed tar member's bytes out of the archive buffer. Sparse
* members cannot be reassembled from a contiguous slice, so reading them throws
* a catchable error instead of returning corrupt data.
*/
function extractTarMember(storage: TarStorage, size: number, memberPath: string): Uint8Array {
if (storage.sparse) {
throw new ToolError(`Archive member '${memberPath}' is a sparse file and cannot be read`);
}
const end = storage.dataOffset + size;
if (end > storage.buffer.length) {
throw new ToolError(`Archive member '${memberPath}' is truncated`);
}
return storage.buffer.subarray(storage.dataOffset, end);
}
function throwUnreadableTarLink(storage: TarLinkStorage, memberPath: string): never {
throw new ToolError(`Archive symlink '${memberPath}' cannot be materialized from target '${storage.targetPath}'`);
}
async function readZipEntries(source: ByteSource): Promise<ArchiveIndexEntry[]> {
const directoryInfo = await readZipCentralDirectoryInfo(source);
const centralDirectory = await source.read(directoryInfo.offset, directoryInfo.offset + directoryInfo.size);
@@ -675,7 +1015,8 @@ export function parseArchivePathCandidates(filePath: string): ArchivePathCandida
/**
* An indexed, read-only view over a single archive. ZIP archives are indexed
* from the central directory and members are inflated on demand; tar archives
* are fully materialized by `Bun.Archive` up front.
* are parsed from one in-memory buffer, members are sliced on demand, and
* directory symlink aliases are traversed lazily.
*/
export class ArchiveReader {
readonly format: ArchiveFormat;
@@ -689,6 +1030,34 @@ export class ArchiveReader {
ensureParentDirectories(this.#entries);
}
#resolveDirectoryAliases(archivePath: string): string {
let resolvedPath = archivePath;
const seen = new Set<string>();
while (true) {
if (seen.has(resolvedPath)) {
throw new ToolError(`Archive path '${archivePath}' crosses a cyclic symlink`);
}
seen.add(resolvedPath);
const parts = resolvedPath.split("/");
let replacement: string | undefined;
for (let end = parts.length; end > 0; end--) {
const prefix = parts.slice(0, end).join("/");
const entry = this.#entries.get(prefix);
if (!entry?.isDirectory || entry.storage?.type !== "tar-link") continue;
const suffix = parts.slice(end).join("/");
replacement = suffix
? entry.storage.targetPath
? `${entry.storage.targetPath}/${suffix}`
: suffix
: entry.storage.targetPath;
break;
}
if (replacement === undefined) return resolvedPath;
resolvedPath = replacement;
}
}
getNode(subPath?: string): ArchiveNode | undefined {
const normalizedPath = normalizeArchiveLookupPath(subPath);
if (normalizedPath === undefined) return undefined;
@@ -696,10 +1065,14 @@ export class ArchiveReader {
return { path: "", isDirectory: true, size: 0 };
}
const entry = this.#entries.get(normalizedPath);
const resolvedPath = this.#resolveDirectoryAliases(normalizedPath);
if (resolvedPath === "") {
return { path: normalizedPath, isDirectory: true, size: 0 };
}
const entry = this.#entries.get(resolvedPath);
if (!entry) return undefined;
return {
path: entry.path,
path: normalizedPath,
isDirectory: entry.isDirectory,
size: entry.size,
mtimeMs: entry.mtimeMs,
@@ -712,8 +1085,9 @@ export class ArchiveReader {
throw new ToolError("Archive path cannot contain '..'");
}
if (normalizedPath) {
const entry = this.#entries.get(normalizedPath);
const resolvedPath = normalizedPath ? this.#resolveDirectoryAliases(normalizedPath) : "";
if (normalizedPath && resolvedPath !== "") {
const entry = this.#entries.get(resolvedPath);
if (!entry) {
throw new ToolError(`Archive path '${normalizedPath}' not found`);
}
@@ -722,22 +1096,23 @@ export class ArchiveReader {
}
}
const prefix = normalizedPath ? `${normalizedPath}/` : "";
const sourcePrefix = resolvedPath ? `${resolvedPath}/` : "";
const children = new Map<string, ArchiveDirectoryEntry>();
for (const entry of this.#entries.values()) {
if (normalizedPath) {
if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue;
if (resolvedPath) {
if (!entry.path.startsWith(sourcePrefix) || entry.path === resolvedPath) continue;
}
const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path;
const relativePath = resolvedPath ? entry.path.slice(sourcePrefix.length) : entry.path;
const nextSegment = relativePath.split("/")[0];
if (!nextSegment) continue;
const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment;
if (children.has(childPath)) continue;
const childEntry = this.#entries.get(childPath);
const sourceChildPath = resolvedPath ? `${resolvedPath}/${nextSegment}` : nextSegment;
const childEntry = this.#entries.get(sourceChildPath);
const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/");
children.set(childPath, {
name: nextSegment,
@@ -759,7 +1134,11 @@ export class ArchiveReader {
throw new ToolError("Archive file path is required");
}
const entry = this.#entries.get(normalizedPath);
const resolvedPath = this.#resolveDirectoryAliases(normalizedPath);
if (resolvedPath === "") {
throw new ToolError(`Archive path '${normalizedPath}' is a directory`);
}
const entry = this.#entries.get(resolvedPath);
if (!entry) {
throw new ToolError(`Archive file '${normalizedPath}' not found`);
}
@@ -775,13 +1154,16 @@ export class ArchiveReader {
);
}
const bytes =
entry.storage.type === "tar"
? await entry.storage.file.bytes()
: await readZipFileBytes(entry.storage, entry.size);
let bytes: Uint8Array;
if (entry.storage.type === "tar") {
bytes = extractTarMember(entry.storage, entry.size, normalizedPath);
} else if (entry.storage.type === "tar-link") {
throwUnreadableTarLink(entry.storage, normalizedPath);
} else {
bytes = await readZipFileBytes(entry.storage, entry.size);
}
return {
path: entry.path,
path: normalizedPath,
isDirectory: false,
size: entry.size,
mtimeMs: entry.mtimeMs,
@@ -812,7 +1194,7 @@ export async function openArchive(source: ArchiveSource): Promise<ArchiveReader>
`Archive is too large to read in memory (${formatBytes(archiveSize)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
);
}
return new ArchiveReader(format, await readTarEntries(await file.bytes()));
return new ArchiveReader(format, readTarEntries(await file.bytes()));
}
const { bytes, format } = source;
@@ -824,7 +1206,7 @@ export async function openArchive(source: ArchiveSource): Promise<ArchiveReader>
`Archive is too large to read in memory (${formatBytes(bytes.byteLength)} > ${formatBytes(MAX_TAR_ARCHIVE_BYTES)} limit)`,
);
}
return new ArchiveReader(format, await readTarEntries(bytes));
return new ArchiveReader(format, readTarEntries(bytes));
}
/** Render the top-level entries of an in-memory archive as one line each. */
@@ -857,9 +1239,9 @@ async function memberToBytes(content: ArchiveMemberContent): Promise<Uint8Array>
/**
* Fully materialize every file member into a `path → content` map: ZIP members
* are inflated in memory, tar members are returned as lazy `File`s. Use this
* when you need every entry (rewrite, extract); for browsing or single-member
* reads prefer `openArchive`, which is lazy for ZIP.
* are inflated in memory, tar members are sliced from the decoded archive
* buffer. Use this when you need every entry (rewrite, extract); for browsing
* or single-member reads prefer `openArchive`, which is lazy for ZIP.
*/
export async function readArchiveEntries(source: ArchiveSource): Promise<Map<string, ArchiveMemberContent>> {
const { bytes, format } = await resolveArchiveBytes(source);
@@ -871,9 +1253,23 @@ export async function readArchiveEntries(source: ArchiveSource): Promise<Map<str
}
return entries;
}
const files = await new Bun.Archive(bytes).files();
for (const [name, file] of files) {
entries.set(name.replace(/\\/g, "/"), file);
for (const entry of readTarEntries(bytes)) {
if (entry.isDirectory) {
if (entry.storage?.type === "tar-link") {
throwUnreadableTarLink(entry.storage, entry.path);
}
continue;
}
if (!entry.storage) {
throw new ToolError(`Archive file '${entry.path}' has no readable storage`);
}
if (entry.storage.type === "tar-link") {
throwUnreadableTarLink(entry.storage, entry.path);
}
if (entry.storage.type !== "tar") {
throw new ToolError(`Archive file '${entry.path}' has invalid tar storage`);
}
entries.set(entry.path, extractTarMember(entry.storage, entry.size, entry.path));
}
return entries;
}
+265 -2
View File
@@ -15,7 +15,7 @@ import { wrapToolWithMetaNotice } from "@oh-my-pi/pi-coding-agent/tools/output-m
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
import * as toolTimeouts from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts";
import { WriteTool } from "@oh-my-pi/pi-coding-agent/tools/write";
import { unzip } from "@oh-my-pi/pi-coding-agent/utils/zip";
import { readArchiveEntries, unzip } from "@oh-my-pi/pi-coding-agent/utils/zip";
import { $which, removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
import { GlobTool } from "../src/tools/glob";
import { DEFAULT_FILE_LIMIT, GrepTool, MULTI_FILE_PER_FILE_MATCHES } from "../src/tools/grep";
@@ -64,6 +64,9 @@ function createFifoOrSkip(fifoPath: string): boolean {
interface ArchiveFixtureEntry {
path: string;
content: string;
prefix?: string;
typeFlag?: "0" | "1" | "2";
linkName?: string;
}
function writeTarString(buffer: Buffer, offset: number, length: number, value: string): void {
@@ -85,13 +88,15 @@ function createTarArchive(entries: ArchiveFixtureEntry[]): Buffer {
const content = Buffer.from(entry.content, "utf-8");
writeTarString(header, 0, 100, entry.path);
if (entry.prefix) writeTarString(header, 345, 155, entry.prefix);
if (entry.linkName) writeTarString(header, 157, 100, entry.linkName);
writeTarOctal(header, 100, 8, 0o644);
writeTarOctal(header, 108, 8, 0);
writeTarOctal(header, 116, 8, 0);
writeTarOctal(header, 124, 12, content.length);
writeTarOctal(header, 136, 12, Math.floor(Date.now() / 1000));
header.fill(0x20, 148, 156);
header[156] = "0".charCodeAt(0);
header[156] = (entry.typeFlag ?? "0").charCodeAt(0);
writeTarString(header, 257, 6, "ustar");
writeTarString(header, 263, 2, "00");
@@ -113,6 +118,67 @@ function createTarArchive(entries: ArchiveFixtureEntry[]): Buffer {
return Buffer.concat(parts);
}
function tarChecksum(header: Buffer): void {
header.fill(0x20, 148, 156);
let checksum = 0;
for (const byte of header) checksum += byte;
header.write(checksum.toString(8).padStart(6, "0"), 148, 6, "ascii");
header[154] = 0;
header[155] = 0x20;
}
function paxRecord(key: string, value: string): Buffer {
const suffix = ` ${key}=${value}\n`;
let length = suffix.length;
while (`${length}`.length + suffix.length !== length) {
length = `${length}`.length + suffix.length;
}
return Buffer.from(`${length}${suffix}`, "utf-8");
}
/**
* Build a GNU 1.0 sparse PAX archive: an `x` extended header carrying the
* user-visible `GNU.sparse.name`/`GNU.sparse.realsize`, followed by a regular
* file header using the internal `GNUSparseFile.NNN` path.
*/
function createSparsePaxTarArchive(realName: string, realSize: number, storedData: Buffer): Buffer {
const paxBody = Buffer.concat([
paxRecord("GNU.sparse.major", "1"),
paxRecord("GNU.sparse.minor", "0"),
paxRecord("GNU.sparse.name", realName),
paxRecord("GNU.sparse.realsize", `${realSize}`),
paxRecord("size", `${storedData.length}`),
]);
const paxHeader = Buffer.alloc(512, 0);
writeTarString(paxHeader, 0, 100, "./PaxHeaders/sparse");
writeTarOctal(paxHeader, 100, 8, 0o644);
writeTarOctal(paxHeader, 124, 12, paxBody.length);
writeTarOctal(paxHeader, 136, 12, Math.floor(Date.now() / 1000));
paxHeader[156] = "x".charCodeAt(0);
writeTarString(paxHeader, 257, 6, "ustar");
writeTarString(paxHeader, 263, 2, "00");
tarChecksum(paxHeader);
const fileHeader = Buffer.alloc(512, 0);
writeTarString(fileHeader, 0, 100, "./GNUSparseFile.0/sparse.bin");
writeTarOctal(fileHeader, 100, 8, 0o644);
writeTarOctal(fileHeader, 124, 12, storedData.length);
writeTarOctal(fileHeader, 136, 12, Math.floor(Date.now() / 1000));
fileHeader[156] = "0".charCodeAt(0);
writeTarString(fileHeader, 257, 6, "ustar");
writeTarString(fileHeader, 263, 2, "00");
tarChecksum(fileHeader);
const parts: Buffer[] = [paxHeader];
const paxRemainder = paxBody.length % 512;
parts.push(paxBody, paxRemainder === 0 ? Buffer.alloc(0) : Buffer.alloc(512 - paxRemainder, 0));
parts.push(fileHeader, storedData);
const dataRemainder = storedData.length % 512;
if (dataRemainder !== 0) parts.push(Buffer.alloc(512 - dataRemainder, 0));
parts.push(Buffer.alloc(1024, 0));
return Buffer.concat(parts);
}
const CRC32_TABLE = (() => {
const table = new Uint32Array(256);
for (let index = 0; index < 256; index++) {
@@ -753,6 +819,203 @@ describe("Coding Agent Tools", () => {
expect(result.details?.isDirectory).toBe(true);
});
it("should read tar.gz members with UTF-8 ustar prefixes", async () => {
const archivePath = path.join(testDir, "unicode-prefix.tar.gz");
const prefix = "bun-da3851e57ae130c5594d0e208a5da5ba8c13edfb/test/js/node/test/fixtures/copy/utf/新建文件夹";
const memberPath = `${prefix}/experimental.json`;
fs.writeFileSync(
archivePath,
zlib.gzipSync(
createTarArchive([
{
path: "experimental.json",
prefix,
content: '{ "type": "module" }',
},
]),
),
);
const rootResult = await readTool.execute("test-call-tar-unicode-prefix-root", { path: archivePath });
expect(getTextOutput(rootResult)).toContain("bun-da3851e57ae130c5594d0e208a5da5ba8c13edfb/");
expect(rootResult.details?.isDirectory).toBe(true);
const memberResult = await readTool.execute("test-call-tar-unicode-prefix-member", {
path: `${archivePath}:${memberPath}`,
});
expect(getTextOutput(memberResult)).toContain('{ "type": "module" }');
});
it("should preserve tar hard-link members", async () => {
const archivePath = path.join(testDir, "hard-link.tar");
fs.writeFileSync(
archivePath,
createTarArchive([
{ path: "pkg/original.txt", content: "shared content\n" },
{ path: "pkg/linked.txt", content: "", typeFlag: "1", linkName: "pkg/original.txt" },
]),
);
const rootResult = await readTool.execute("test-call-tar-hard-link-root", { path: `${archivePath}:pkg` });
expect(getTextOutput(rootResult)).toContain("linked.txt");
const linkedResult = await readTool.execute("test-call-tar-hard-link-member", {
path: `${archivePath}:pkg/linked.txt`,
});
expect(getTextOutput(linkedResult)).toContain("shared content");
const entries = await readArchiveEntries(archivePath);
const linkedContent = entries.get("pkg/linked.txt");
if (!(linkedContent instanceof Uint8Array)) {
throw new Error("Expected hard-link content to materialize as bytes");
}
expect(new TextDecoder().decode(linkedContent)).toBe("shared content\n");
});
it("should preserve safe relative tar file symlinks", async () => {
const archivePath = path.join(testDir, "file-symlink.tar");
fs.writeFileSync(
archivePath,
createTarArchive([
{ path: "pkg/lib/tool.js", content: "export const linked = true;\n" },
{ path: "pkg/bin/tool", content: "", typeFlag: "2", linkName: "../lib/tool.js" },
]),
);
const linkedResult = await readTool.execute("test-call-tar-symlink-member", {
path: `${archivePath}:pkg/bin/tool`,
});
expect(getTextOutput(linkedResult)).toContain("export const linked = true");
const entries = await readArchiveEntries(archivePath);
const linkedContent = entries.get("pkg/bin/tool");
if (!(linkedContent instanceof Uint8Array)) {
throw new Error("Expected symlink content to materialize as bytes");
}
expect(new TextDecoder().decode(linkedContent)).toBe("export const linked = true;\n");
});
it("should resolve directory symlinks lazily without materializing subtrees", async () => {
const archivePath = path.join(testDir, "directory-symlinks.tar");
fs.writeFileSync(
archivePath,
createTarArchive([
{ path: "pkg/lib/tool.js", content: "export const linked = true;\n" },
{ path: "pkg/lib/extra.js", content: "export const extra = true;\n" },
{ path: "pkg/current-a", content: "", typeFlag: "2", linkName: "lib" },
{ path: "pkg/current-b", content: "", typeFlag: "2", linkName: "lib" },
{ path: "pkg/current-c", content: "", typeFlag: "2", linkName: "lib" },
]),
);
const linkedResult = await readTool.execute("test-call-tar-directory-symlink-member", {
path: `${archivePath}:pkg/current-a/tool.js`,
});
expect(getTextOutput(linkedResult)).toContain("export const linked = true");
const directoryResult = await readTool.execute("test-call-tar-directory-symlink-directory", {
path: `${archivePath}:pkg/current-b`,
});
expect(getTextOutput(directoryResult)).toContain("extra.js");
await expect(readArchiveEntries(archivePath)).rejects.toThrow(/cannot be materialized/);
});
it("should resolve tar symlinks whose target is the archive root", async () => {
const archivePath = path.join(testDir, "root-symlinks.tar");
fs.writeFileSync(
archivePath,
createTarArchive([
{ path: "top.txt", content: "top level\n" },
{ path: "dir/inner.txt", content: "inner\n" },
// `current -> .` and `dir/up -> ..` both normalize to the archive root.
{ path: "current", content: "", typeFlag: "2", linkName: "." },
{ path: "dir/up", content: "", typeFlag: "2", linkName: ".." },
]),
);
const currentNode = await readTool.execute("test-call-tar-root-symlink-current", {
path: `${archivePath}:current/top.txt`,
});
expect(getTextOutput(currentNode)).toContain("top level");
const upNode = await readTool.execute("test-call-tar-root-symlink-up", {
path: `${archivePath}:dir/up/top.txt`,
});
expect(getTextOutput(upNode)).toContain("top level");
});
it("should list dangling tar symlinks but reject their materialization", async () => {
const archivePath = path.join(testDir, "dangling-symlink.tar");
fs.writeFileSync(
archivePath,
createTarArchive([{ path: "pkg/dangling", content: "", typeFlag: "2", linkName: "missing-target" }]),
);
const rootResult = await readTool.execute("test-call-tar-dangling-symlink-root", {
path: `${archivePath}:pkg`,
});
expect(getTextOutput(rootResult)).toContain("dangling");
await expect(
readTool.execute("test-call-tar-dangling-symlink-member", {
path: `${archivePath}:pkg/dangling`,
}),
).rejects.toThrow(/cannot be materialized/);
await expect(readArchiveEntries(archivePath)).rejects.toThrow(/cannot be materialized/);
});
it("should surface GNU sparse PAX names and reject sparse reads", async () => {
const archivePath = path.join(testDir, "sparse-pax.tar");
fs.writeFileSync(
archivePath,
createSparsePaxTarArchive("data/sparse.bin", 1048576, Buffer.from("sparse-map\n")),
);
// The listing must show the real GNU.sparse.name, not the internal
// GNUSparseFile.NNN path.
const rootResult = await readTool.execute("test-call-tar-sparse-root", { path: `${archivePath}:data` });
expect(getTextOutput(rootResult)).toContain("sparse.bin");
expect(getTextOutput(rootResult)).not.toContain("GNUSparseFile");
// Reading the real member name resolves the entry and rejects it as
// sparse (a catchable error), rather than reporting it missing.
await expect(
readTool.execute("test-call-tar-sparse-member", { path: `${archivePath}:data/sparse.bin` }),
).rejects.toThrow(/sparse file and cannot be read/);
});
it("should reject a truncated tar member while indexing", async () => {
const archivePath = path.join(testDir, "truncated.tar");
// A full, valid archive declares 2048 bytes for `big.txt`; slicing the
// payload mid-member leaves the header's declared size pointing past EOF.
const complete = createTarArchive([{ path: "big.txt", content: "A".repeat(2048) }]);
fs.writeFileSync(archivePath, complete.subarray(0, 512 + 256));
await expect(readTool.execute("test-call-tar-truncated", { path: archivePath })).rejects.toThrow(/truncated/);
});
it("should reject a tar truncated before its terminating zero block", async () => {
const archivePath = path.join(testDir, "unterminated.tar");
const complete = createTarArchive([{ path: "complete.txt", content: "complete member\n" }]);
fs.writeFileSync(archivePath, complete.subarray(0, complete.length - 1024));
await expect(readTool.execute("test-call-tar-unterminated", { path: archivePath })).rejects.toThrow(
/missing terminating zero block/,
);
});
it("should reject a gzip payload that is not a tar archive", async () => {
// `sniffArchiveFormat` classifies any gzip magic as tar.gz, so a plain
// `.txt.gz` (decompressed payload shorter than one 512-byte tar block)
// must raise a catchable error instead of listing an empty directory.
const archivePath = path.join(testDir, "note.tar.gz");
fs.writeFileSync(archivePath, zlib.gzipSync(Buffer.from("hello world\n")));
await expect(readTool.execute("test-call-gzip-non-tar", { path: archivePath })).rejects.toThrow(
/not a valid tar archive/i,
);
});
it("should list archive subdirectories", async () => {
const archivePath = path.join(testDir, "fixture.zip");
fs.writeFileSync(