diff --git a/bun.lock b/bun.lock index daf9cf3f0..8f3e476c8 100644 --- a/bun.lock +++ b/bun.lock @@ -69,6 +69,7 @@ "ajv": "^8.18", "chalk": "^5.6", "diff": "^8.0", + "fflate": "0.8.2", "handlebars": "^4.7", "linkedom": "^0.18", "puppeteer": "^24.37", @@ -647,6 +648,8 @@ "fetch-blob": ["fetch-blob@3.2.0", "", { "dependencies": { "node-domexception": "^1.0.0", "web-streams-polyfill": "^3.0.3" } }, "sha512-7yAQpD2UMJzLi1Dqv7qFYnPbaPx7ZfFK6PiIxQ4PfkGPyNyl2Ugx+a/umUonmKqjhM4DnfbMvdX6otXq83soQQ=="], + "fflate": ["fflate@0.8.2", "", {}, "sha512-cPJU47OaAoCbg0pBvzsgpTPhmhqI5eJjh/JIu8tPj5q+T7iLvW/JAYUqmE7KOB4R1ZyEhzBaIQpQpardBF5z8A=="], + "file-stream-rotator": ["file-stream-rotator@0.6.1", "", { "dependencies": { "moment": "^2.29.1" } }, "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ=="], "fn.name": ["fn.name@1.1.0", "", {}, "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw=="], diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c64e2a731..2b84ec015 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,11 +1,17 @@ # Changelog ## [Unreleased] - ### Added +- Added support for reading files from `.tar`, `.tar.gz`, `.tgz`, and `.zip` archives using virtual subpaths like `archive.ext:path/to/file` +- Added ability to list archive contents and navigate subdirectories within supported archive formats +- Added archive-aware `read` support for `.tar`, `.tar.gz`, `.tgz`, and `.zip`, including virtual subpaths like `archive.ext:path/to/file` - Added `/tools` slash command to show the tools currently visible to the agent in the interactive session +### Changed + +- Updated `read` tool documentation to reflect archive support and usage patterns + ## [13.17.2] - 2026-04-01 ### Added diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index a293225ae..3b8a1003c 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -53,6 +53,7 @@ "ajv": "^8.18", "chalk": "^5.6", "diff": "^8.0", + "fflate": "0.8.2", "handlebars": "^4.7", "linkedom": "^0.18", "puppeteer": "^24.37", diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index bc641d21c..e42d0c591 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -1,4 +1,4 @@ -Reads files from local filesystem or internal URLs. +Reads files from local filesystem, supported archives, or internal URLs. - Reads up to {{DEFAULT_LIMIT}} lines default @@ -12,17 +12,20 @@ Reads files from local filesystem or internal URLs. {{/if}} - Supports images (PNG, JPG) and PDFs - For directories, returns formatted listing with modification times +- Supports `.tar`, `.tar.gz`, `.tgz`, and `.zip` archives +- Use `archive.ext:path/inside/archive` to read or list archive contents - Parallelize reads when exploring related files -- Returns file content as text; images return visual content; PDFs return extracted text +- Returns file content as text; images return visual content; PDFs return extracted text; archive roots behave like directories - Missing files: returns closest filename matches for correction - You **MUST** use `read` instead of bash for ALL file reading: `cat`, `head`, `tail`, `less`, `more` are FORBIDDEN. - You **MUST** use `read(path="dir/")` instead of `ls dir/` for directory listings. +- You **MUST** use `read(path="archive.zip:path/to/file")` instead of shelling out to `tar` or `unzip` for supported archive reads. - You **MUST** always include the `path` parameter — NEVER call `read` with empty arguments `{}`. - When reading specific line ranges, use `offset` and `limit`: `read(path="file", offset=50, limit=100)` not `cat -n file | sed`. diff --git a/packages/coding-agent/src/tools/archive-reader.ts b/packages/coding-agent/src/tools/archive-reader.ts new file mode 100644 index 000000000..e10373ba8 --- /dev/null +++ b/packages/coding-agent/src/tools/archive-reader.ts @@ -0,0 +1,309 @@ +import { unzipSync } from "fflate"; +import { ToolError } from "./tool-errors"; + +export type ArchiveFormat = "zip" | "tar" | "tar.gz"; + +export interface ArchivePathCandidate { + archivePath: string; + subPath: string; +} + +export interface ArchiveNode { + path: string; + isDirectory: boolean; + size: number; + mtimeMs?: number; +} + +export interface ArchiveDirectoryEntry extends ArchiveNode { + name: string; +} + +export interface ExtractedArchiveFile extends ArchiveNode { + bytes: Uint8Array; +} + +interface TarStorage { + type: "tar"; + file: File; +} + +interface ZipStorage { + type: "zip"; + bytes: Uint8Array; +} + +type EntryStorage = TarStorage | ZipStorage; + +interface ArchiveIndexEntry extends ArchiveNode { + storage?: EntryStorage; +} + +function normalizeArchiveLookupPath(rawPath?: string): string | undefined { + if (!rawPath) return ""; + + const parts = rawPath.replace(/\\/g, "/").split("/"); + const normalizedParts: string[] = []; + for (const part of parts) { + if (!part || part === ".") continue; + if (part === "..") return undefined; + normalizedParts.push(part); + } + + return normalizedParts.join("/"); +} + +function normalizeArchiveEntryPath(rawPath: string): string | undefined { + const parts = rawPath.replace(/\\/g, "/").split("/"); + const normalizedParts: string[] = []; + for (const part of parts) { + if (!part || part === ".") continue; + if (part === "..") return undefined; + normalizedParts.push(part); + } + + if (normalizedParts.length === 0) return undefined; + return normalizedParts.join("/"); +} + +function isArchiveDirectoryName(rawPath: string): boolean { + return rawPath.endsWith("/") || rawPath.endsWith("\\"); +} + +function upsertArchiveEntry(map: Map, entry: ArchiveIndexEntry): void { + const existing = map.get(entry.path); + if (!existing) { + map.set(entry.path, entry); + return; + } + + if (existing.isDirectory && !entry.isDirectory) { + map.set(entry.path, entry); + return; + } + + if (!existing.isDirectory && entry.isDirectory) { + return; + } + + map.set(entry.path, { + ...existing, + size: existing.size || entry.size, + mtimeMs: existing.mtimeMs ?? entry.mtimeMs, + storage: existing.storage ?? entry.storage, + }); +} + +function ensureParentDirectories(map: Map): void { + for (const entry of [...map.values()]) { + const parts = entry.path.split("/"); + const stop = parts.length - 1; + for (let index = 1; index <= stop; index++) { + const dirPath = parts.slice(0, index).join("/"); + if (!dirPath || map.has(dirPath)) continue; + map.set(dirPath, { + path: dirPath, + isDirectory: true, + size: 0, + }); + } + } +} + +function getArchiveFormatFromPath(filePath: string): ArchiveFormat | undefined { + const normalized = filePath.toLowerCase(); + if (normalized.endsWith(".tar.gz") || normalized.endsWith(".tgz")) return "tar.gz"; + if (normalized.endsWith(".tar")) return "tar"; + if (normalized.endsWith(".zip")) return "zip"; + return undefined; +} + +async function readTarEntries(bytes: Uint8Array): Promise { + let archive: Bun.Archive; + try { + archive = new Bun.Archive(bytes); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } + + let files: Map; + try { + files = await archive.files(); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } + + const entries: ArchiveIndexEntry[] = []; + for (const [rawPath, file] of files) { + const normalizedPath = normalizeArchiveEntryPath(rawPath); + if (!normalizedPath) continue; + const mtimeMs = file.lastModified > 0 ? file.lastModified : undefined; + entries.push({ + path: normalizedPath, + isDirectory: false, + size: file.size, + mtimeMs, + storage: { type: "tar", file }, + }); + } + + return entries; +} + +function readZipEntries(bytes: Uint8Array): ArchiveIndexEntry[] { + let files: Record; + try { + files = unzipSync(bytes); + } catch (error) { + throw new ToolError(error instanceof Error ? error.message : String(error)); + } + + const entries: ArchiveIndexEntry[] = []; + for (const [rawPath, fileBytes] of Object.entries(files)) { + const normalizedPath = normalizeArchiveEntryPath(rawPath); + if (!normalizedPath) continue; + const isDirectory = isArchiveDirectoryName(rawPath); + entries.push({ + path: normalizedPath, + isDirectory, + size: isDirectory ? 0 : fileBytes.byteLength, + storage: isDirectory ? undefined : { type: "zip", bytes: fileBytes }, + }); + } + + return entries; +} + +export function parseArchivePathCandidates(filePath: string): ArchivePathCandidate[] { + const normalized = filePath.replace(/\\/g, "/"); + const pattern = /\.(?:tar\.gz|tgz|zip|tar)(?=(?::|$))/gi; + const seen = new Set(); + const candidates: ArchivePathCandidate[] = []; + + let match: RegExpExecArray | null; + while ((match = pattern.exec(normalized)) !== null) { + const end = match.index + match[0].length; + const archivePath = filePath.slice(0, end); + const subPath = normalized.slice(end).replace(/^:+/, ""); + const key = `${archivePath}\0${subPath}`; + if (seen.has(key)) continue; + seen.add(key); + candidates.push({ archivePath, subPath }); + } + + return candidates.sort((left, right) => right.archivePath.length - left.archivePath.length); +} + +export class ArchiveReader { + readonly format: ArchiveFormat; + #entries = new Map(); + + constructor(format: ArchiveFormat, entries: ArchiveIndexEntry[]) { + this.format = format; + for (const entry of entries) { + upsertArchiveEntry(this.#entries, entry); + } + ensureParentDirectories(this.#entries); + } + + getNode(subPath?: string): ArchiveNode | undefined { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (normalizedPath === undefined) return undefined; + if (normalizedPath === "") { + return { path: "", isDirectory: true, size: 0 }; + } + + const entry = this.#entries.get(normalizedPath); + if (!entry) return undefined; + return { + path: entry.path, + isDirectory: entry.isDirectory, + size: entry.size, + mtimeMs: entry.mtimeMs, + }; + } + + listDirectory(subPath?: string): ArchiveDirectoryEntry[] { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (normalizedPath === undefined) { + throw new ToolError("Archive path cannot contain '..'"); + } + + if (normalizedPath) { + const entry = this.#entries.get(normalizedPath); + if (!entry) { + throw new ToolError(`Archive path '${normalizedPath}' not found`); + } + if (!entry.isDirectory) { + throw new ToolError(`Archive path '${normalizedPath}' is not a directory`); + } + } + + const prefix = normalizedPath ? `${normalizedPath}/` : ""; + const children = new Map(); + + for (const entry of this.#entries.values()) { + if (normalizedPath) { + if (!entry.path.startsWith(prefix) || entry.path === normalizedPath) continue; + } + + const relativePath = normalizedPath ? entry.path.slice(prefix.length) : entry.path; + const nextSegment = relativePath.split("/")[0]; + if (!nextSegment) continue; + + const childPath = normalizedPath ? `${normalizedPath}/${nextSegment}` : nextSegment; + if (children.has(childPath)) continue; + + const childEntry = this.#entries.get(childPath); + const isDirectory = childEntry?.isDirectory ?? relativePath.includes("/"); + children.set(childPath, { + name: nextSegment, + path: childPath, + isDirectory, + size: isDirectory ? 0 : (childEntry?.size ?? entry.size), + mtimeMs: childEntry?.mtimeMs ?? entry.mtimeMs, + }); + } + + return [...children.values()].sort((left, right) => left.name.toLowerCase().localeCompare(right.name.toLowerCase())); + } + + async readFile(subPath: string): Promise { + const normalizedPath = normalizeArchiveLookupPath(subPath); + if (!normalizedPath) { + throw new ToolError("Archive file path is required"); + } + + const entry = this.#entries.get(normalizedPath); + if (!entry) { + throw new ToolError(`Archive file '${normalizedPath}' not found`); + } + if (entry.isDirectory) { + throw new ToolError(`Archive path '${normalizedPath}' is a directory`); + } + if (!entry.storage) { + throw new ToolError(`Archive file '${normalizedPath}' has no readable storage`); + } + + const bytes = entry.storage.type === "tar" ? await entry.storage.file.bytes() : entry.storage.bytes; + + return { + path: entry.path, + isDirectory: false, + size: entry.size, + mtimeMs: entry.mtimeMs, + bytes, + }; + } +} + +export async function openArchive(filePath: string): Promise { + const format = getArchiveFormatFromPath(filePath); + if (!format) { + throw new ToolError(`Unsupported archive format: ${filePath}`); + } + + const bytes = await Bun.file(filePath).bytes(); + const entries = format === "zip" ? readZipEntries(bytes) : await readTarEntries(bytes); + return new ArchiveReader(format, entries); +} diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 39179eec7..2d7e2d134 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -1,5 +1,5 @@ import * as fs from "node:fs/promises"; -import path from "node:path"; +import * as path from "node:path"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; import { glob } from "@oh-my-pi/pi-natives"; @@ -26,6 +26,7 @@ import { import { renderCodeCell, renderStatusLine } from "../tui"; import { CachedOutputBlock } from "../tui/output-block"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; +import { ArchiveReader, openArchive, parseArchivePathCandidates } from "./archive-reader"; import { ImageInputTooLargeError, loadImageInput, @@ -340,6 +341,25 @@ async function convertWithMarkitdown( return { content: "", ok: false, error: result.stderr.trim() || "Conversion failed" }; } +function decodeUtf8Text(bytes: Uint8Array): string | null { + for (const byte of bytes) { + if (byte === 0) return null; + } + + try { + return new TextDecoder("utf-8", { fatal: true }).decode(bytes); + } catch { + return null; + } +} + +function prependSuffixResolutionNotice(text: string, suffixResolution?: { from: string; to: string }): string { + if (!suffixResolution) return text; + + const notice = `[Path '${suffixResolution.from}' not found; resolved to '${suffixResolution.to}' via suffix match]`; + return text ? `${notice}\n${text}` : notice; +} + const readSchema = Type.Object({ path: Type.String({ description: "Path to the file to read (relative or absolute)" }), offset: Type.Optional(Type.Number({ description: "Line number to start reading from (1-indexed)" })), @@ -358,6 +378,12 @@ export interface ReadToolDetails { type ReadParams = ReadToolInput; +interface ResolvedArchiveReadPath { + absolutePath: string; + archiveSubPath: string; + suffixResolution?: { from: string; to: string }; +} + /** * Read tool implementation. * @@ -392,6 +418,246 @@ export class ReadTool implements AgentTool { }); } + async #resolveArchiveReadPath(readPath: string, signal?: AbortSignal): Promise { + const candidates = parseArchivePathCandidates(readPath); + for (const candidate of candidates) { + let absolutePath = resolveReadPath(candidate.archivePath, this.session.cwd); + let suffixResolution: { from: string; to: string } | undefined; + + try { + const stat = await Bun.file(absolutePath).stat(); + if (stat.isDirectory()) continue; + return { + absolutePath, + archiveSubPath: candidate.archivePath === readPath ? "" : candidate.subPath, + suffixResolution, + }; + } catch (error) { + if (!isNotFoundError(error) || isRemoteMountPath(absolutePath)) continue; + + const suffixMatch = await findUniqueSuffixMatch(candidate.archivePath, this.session.cwd, signal); + if (!suffixMatch) continue; + + try { + const retryStat = await Bun.file(suffixMatch.absolutePath).stat(); + if (retryStat.isDirectory()) continue; + + absolutePath = suffixMatch.absolutePath; + suffixResolution = { from: candidate.archivePath, to: suffixMatch.displayPath }; + return { + absolutePath, + archiveSubPath: candidate.archivePath === readPath ? "" : candidate.subPath, + suffixResolution, + }; + } catch (retryError) { + if (!isNotFoundError(retryError)) { + throw retryError; + } + } + } + } + + return null; + } + + #buildInMemoryTextResult( + text: string, + offset: number | undefined, + limit: number | undefined, + options: { + details?: ReadToolDetails; + sourcePath?: string; + sourceInternal?: string; + entityLabel: string; + }, + ): AgentToolResult { + const displayMode = resolveFileDisplayMode(this.session); + const details = options.details ?? {}; + const allLines = text.split("\n"); + const totalLines = allLines.length; + const startLine = offset ? Math.max(0, offset - 1) : 0; + const startLineDisplay = startLine + 1; + + const resultBuilder = toolResult(details); + if (options.sourcePath) { + resultBuilder.sourcePath(options.sourcePath); + } + if (options.sourceInternal) { + resultBuilder.sourceInternal(options.sourceInternal); + } + + if (startLine >= allLines.length) { + const suggestion = + allLines.length === 0 + ? `The ${options.entityLabel} is empty.` + : `Use offset=1 to read from the start, or offset=${allLines.length} to read the last line.`; + return resultBuilder + .text(`Offset ${offset} is beyond end of ${options.entityLabel} (${allLines.length} lines total). ${suggestion}`) + .done(); + } + + const endLine = limit !== undefined ? Math.min(startLine + limit, allLines.length) : allLines.length; + const selectedContent = allLines.slice(startLine, endLine).join("\n"); + const userLimitedLines = limit !== undefined ? endLine - startLine : undefined; + const truncation = truncateHead(selectedContent); + + const shouldAddHashLines = displayMode.hashLines; + const shouldAddLineNumbers = shouldAddHashLines ? false : displayMode.lineNumbers; + const formatText = (content: string, startNum: number): string => + formatTextWithMode(content, startNum, shouldAddHashLines, shouldAddLineNumbers); + + let outputText: string; + let truncationInfo: + | { result: TruncationResult; options: { direction: "head"; startLine?: number; totalFileLines?: number } } + | undefined; + + if (truncation.firstLineExceedsLimit) { + const firstLine = allLines[startLine] ?? ""; + const firstLineBytes = Buffer.byteLength(firstLine, "utf-8"); + const snippet = truncateHeadBytes(firstLine, DEFAULT_MAX_BYTES); + + if (shouldAddHashLines) { + outputText = `[Line ${startLineDisplay} is ${formatBytes( + firstLineBytes, + )}, exceeds ${formatBytes(DEFAULT_MAX_BYTES)} limit. Hashline output requires full lines; cannot compute hashes for a truncated preview.]`; + } else { + outputText = formatText(snippet.text, startLineDisplay); + } + + if (snippet.text.length === 0) { + outputText = `[Line ${startLineDisplay} is ${formatBytes( + firstLineBytes, + )}, exceeds ${formatBytes(DEFAULT_MAX_BYTES)} limit. Unable to display a valid UTF-8 snippet.]`; + } + + details.truncation = truncation; + truncationInfo = { + result: truncation, + options: { direction: "head", startLine: startLineDisplay, totalFileLines: totalLines }, + }; + } else if (truncation.truncated) { + outputText = formatText(truncation.content, startLineDisplay); + details.truncation = truncation; + truncationInfo = { + result: truncation, + options: { direction: "head", startLine: startLineDisplay, totalFileLines: totalLines }, + }; + } else if (userLimitedLines !== undefined && startLine + userLimitedLines < allLines.length) { + const remaining = allLines.length - (startLine + userLimitedLines); + const nextOffset = startLine + userLimitedLines + 1; + + outputText = formatText(selectedContent, startLineDisplay); + outputText += `\n\n[${remaining} more lines in ${options.entityLabel}. Use offset=${nextOffset} to continue]`; + } else { + outputText = formatText(truncation.content, startLineDisplay); + } + + resultBuilder.text(outputText); + if (truncationInfo) { + resultBuilder.truncation(truncationInfo.result, truncationInfo.options); + } + return resultBuilder.done(); + } + + async #readArchiveDirectory( + archive: ArchiveReader, + archivePath: string, + subPath: string, + limit: number | undefined, + details: ReadToolDetails, + signal?: AbortSignal, + ): Promise> { + const DEFAULT_LIMIT = 500; + const effectiveLimit = limit ?? DEFAULT_LIMIT; + const entries = archive.listDirectory(subPath); + + const listLimit = applyListLimit(entries, { limit: effectiveLimit }); + const limitedEntries = listLimit.items; + const limitMeta = listLimit.meta; + + const results: string[] = []; + for (const entry of limitedEntries) { + throwIfAborted(signal); + if (entry.isDirectory) { + results.push(`${entry.name}/`); + continue; + } + + const sizeSuffix = entry.size > 0 ? ` (${formatBytes(entry.size)})` : ""; + results.push(`${entry.name}${sizeSuffix}`); + } + + const output = results.length > 0 ? results.join("\n") : "(empty archive directory)"; + const text = prependSuffixResolutionNotice(output, details.suffixResolution); + const truncation = truncateHead(text, { maxLines: Number.MAX_SAFE_INTEGER }); + const directoryDetails: ReadToolDetails = { ...details, isDirectory: true }; + const resultBuilder = toolResult(directoryDetails).text(truncation.content); + resultBuilder.sourcePath(archivePath).limits({ resultLimit: limitMeta.resultLimit?.reached }); + if (truncation.truncated) { + directoryDetails.truncation = truncation; + resultBuilder.truncation(truncation, { direction: "head" }); + } + return resultBuilder.done(); + } + + async #readArchive( + readPath: string, + offset: number | undefined, + limit: number | undefined, + resolvedArchivePath: ResolvedArchiveReadPath, + signal?: AbortSignal, + ): Promise> { + throwIfAborted(signal); + const archive = await openArchive(resolvedArchivePath.absolutePath); + throwIfAborted(signal); + + const details: ReadToolDetails = { + resolvedPath: resolvedArchivePath.absolutePath, + suffixResolution: resolvedArchivePath.suffixResolution, + }; + + const node = archive.getNode(resolvedArchivePath.archiveSubPath); + if (!node) { + throw new ToolError(`Path '${readPath}' not found inside archive`); + } + + if (node.isDirectory) { + return this.#readArchiveDirectory( + archive, + resolvedArchivePath.absolutePath, + resolvedArchivePath.archiveSubPath, + limit, + details, + signal, + ); + } + + const entry = await archive.readFile(resolvedArchivePath.archiveSubPath); + const text = decodeUtf8Text(entry.bytes); + if (text === null) { + return toolResult(details) + .text( + prependSuffixResolutionNotice( + `[Cannot read binary archive entry '${entry.path}' (${formatBytes(entry.size)})]`, + resolvedArchivePath.suffixResolution, + ), + ) + .sourcePath(resolvedArchivePath.absolutePath) + .done(); + } + + const result = this.#buildInMemoryTextResult(text, offset, limit, { + details, + sourcePath: resolvedArchivePath.absolutePath, + entityLabel: "archive entry", + }); + const firstText = result.content.find((content): content is TextContent => content.type === "text"); + if (firstText) { + firstText.text = prependSuffixResolutionNotice(firstText.text, resolvedArchivePath.suffixResolution); + } + return result; + } + async execute( _toolCallId: string, params: ReadParams, @@ -409,6 +675,11 @@ export class ReadTool implements AgentTool { return this.#handleInternalUrl(readPath, offset, limit); } + const archivePath = await this.#resolveArchiveReadPath(readPath, signal); + if (archivePath) { + return this.#readArchive(readPath, offset, limit, archivePath, signal); + } + let absolutePath = resolveReadPath(readPath, this.session.cwd); let suffixResolution: { from: string; to: string } | undefined; diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 5e9c9d01c..4a1d63d2d 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; +import * as zlib from "node:zlib"; import type { AgentToolContext } from "@oh-my-pi/pi-agent-core"; import { DEFAULT_BASH_INTERCEPTOR_RULES, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { EditTool } from "@oh-my-pi/pi-coding-agent/patch"; @@ -44,6 +45,128 @@ function createFifoOrSkip(fifoPath: string): boolean { return true; } +interface ArchiveFixtureEntry { + path: string; + content: string; +} + +function writeTarString(buffer: Buffer, offset: number, length: number, value: string): void { + const valueBuffer = Buffer.from(value, "utf-8"); + valueBuffer.copy(buffer, offset, 0, Math.min(valueBuffer.length, length)); +} + +function writeTarOctal(buffer: Buffer, offset: number, length: number, value: number): void { + const octal = value.toString(8).padStart(length - 1, "0"); + buffer.write(octal, offset, length - 1, "ascii"); + buffer[offset + length - 1] = 0; +} + +function createTarArchive(entries: ArchiveFixtureEntry[]): Buffer { + const parts: Buffer[] = []; + + for (const entry of entries) { + const header = Buffer.alloc(512, 0); + const content = Buffer.from(entry.content, "utf-8"); + + writeTarString(header, 0, 100, entry.path); + writeTarOctal(header, 100, 8, 0o644); + writeTarOctal(header, 108, 8, 0); + writeTarOctal(header, 116, 8, 0); + writeTarOctal(header, 124, 12, content.length); + writeTarOctal(header, 136, 12, Math.floor(Date.now() / 1000)); + header.fill(0x20, 148, 156); + header[156] = "0".charCodeAt(0); + writeTarString(header, 257, 6, "ustar"); + writeTarString(header, 263, 2, "00"); + + let checksum = 0; + for (const byte of header) checksum += byte; + const checksumText = checksum.toString(8).padStart(6, "0"); + header.write(checksumText, 148, 6, "ascii"); + header[154] = 0; + header[155] = 0x20; + + parts.push(header, content); + const remainder = content.length % 512; + if (remainder !== 0) { + parts.push(Buffer.alloc(512 - remainder, 0)); + } + } + + parts.push(Buffer.alloc(1024, 0)); + return Buffer.concat(parts); +} + +const CRC32_TABLE = (() => { + const table = new Uint32Array(256); + for (let index = 0; index < 256; index++) { + let value = index; + for (let bit = 0; bit < 8; bit++) { + value = (value & 1) !== 0 ? 0xedb88320 ^ (value >>> 1) : value >>> 1; + } + table[index] = value >>> 0; + } + return table; +})(); + +function crc32(bytes: Uint8Array): number { + let value = 0xffffffff; + for (const byte of bytes) { + value = CRC32_TABLE[(value ^ byte) & 0xff]! ^ (value >>> 8); + } + return (value ^ 0xffffffff) >>> 0; +} + +function createZipArchive(entries: ArchiveFixtureEntry[]): Buffer { + const localParts: Buffer[] = []; + const centralParts: Buffer[] = []; + let localOffset = 0; + + for (const entry of entries) { + const pathBuffer = Buffer.from(entry.path.replace(/\\/g, "/"), "utf-8"); + const content = Buffer.from(entry.content, "utf-8"); + const compressed = zlib.deflateRawSync(content); + const checksum = crc32(content); + + const localHeader = Buffer.alloc(30, 0); + localHeader.writeUInt32LE(0x04034b50, 0); + localHeader.writeUInt16LE(20, 4); + localHeader.writeUInt16LE(0x0800, 6); + localHeader.writeUInt16LE(8, 8); + localHeader.writeUInt32LE(checksum, 14); + localHeader.writeUInt32LE(compressed.length, 18); + localHeader.writeUInt32LE(content.length, 22); + localHeader.writeUInt16LE(pathBuffer.length, 26); + + localParts.push(localHeader, pathBuffer, compressed); + + const centralHeader = Buffer.alloc(46, 0); + centralHeader.writeUInt32LE(0x02014b50, 0); + centralHeader.writeUInt16LE(20, 4); + centralHeader.writeUInt16LE(20, 6); + centralHeader.writeUInt16LE(0x0800, 8); + centralHeader.writeUInt16LE(8, 10); + centralHeader.writeUInt32LE(checksum, 16); + centralHeader.writeUInt32LE(compressed.length, 20); + centralHeader.writeUInt32LE(content.length, 24); + centralHeader.writeUInt16LE(pathBuffer.length, 28); + centralHeader.writeUInt32LE(localOffset, 42); + + centralParts.push(centralHeader, pathBuffer); + localOffset += localHeader.length + pathBuffer.length + compressed.length; + } + + const centralDirectory = Buffer.concat(centralParts); + const endOfCentralDirectory = Buffer.alloc(22, 0); + endOfCentralDirectory.writeUInt32LE(0x06054b50, 0); + endOfCentralDirectory.writeUInt16LE(entries.length, 8); + endOfCentralDirectory.writeUInt16LE(entries.length, 10); + endOfCentralDirectory.writeUInt32LE(centralDirectory.length, 12); + endOfCentralDirectory.writeUInt32LE(localOffset, 16); + + return Buffer.concat([...localParts, centralDirectory, endOfCentralDirectory]); +} + let artifactCounter = 0; function createTestToolSession(cwd: string, settings: Settings = Settings.isolated()): ToolSession { const sessionFile = path.join(cwd, "session.jsonl"); @@ -252,6 +375,89 @@ describe("Coding Agent Tools", () => { expect(result.details?.truncation?.outputLines).toBe(defaultLimit); }); + it("should treat .tar archives like directories", async () => { + const archivePath = path.join(testDir, "fixture.tar"); + fs.writeFileSync( + archivePath, + createTarArchive([ + { path: "pkg/README.md", content: "# Tar README\nLine 2\n" }, + { path: "pkg/src/index.ts", content: "export const tarValue = 1;\n" }, + { path: "top.txt", content: "top level\n" }, + ]), + ); + + const result = await readTool.execute("test-call-tar-root", { path: archivePath }); + const output = getTextOutput(result); + + expect(output).toContain("pkg/"); + expect(output).toContain("top.txt"); + expect(result.details?.isDirectory).toBe(true); + }); + + it("should list archive subdirectories", async () => { + const archivePath = path.join(testDir, "fixture.zip"); + fs.writeFileSync( + archivePath, + createZipArchive([ + { path: "pkg/README.md", content: "# Zip README\n" }, + { path: "pkg/src/index.ts", content: "export const zipValue = 2;\n" }, + { path: "pkg/src/util.ts", content: "export const utilValue = 3;\n" }, + ]), + ); + + const result = await readTool.execute("test-call-zip-dir", { path: `${archivePath}:pkg/src` }); + const output = getTextOutput(result); + + expect(output).toContain("index.ts"); + expect(output).toContain("util.ts"); + expect(result.details?.isDirectory).toBe(true); + }); + + for (const archiveCase of [ + { + label: ".tar", + path: "fixture-subpath.tar", + create: (entries: ArchiveFixtureEntry[]) => createTarArchive(entries), + }, + { + label: ".tar.gz", + path: "fixture-subpath.tar.gz", + create: (entries: ArchiveFixtureEntry[]) => zlib.gzipSync(createTarArchive(entries)), + }, + { + label: ".tgz", + path: "fixture-subpath.tgz", + create: (entries: ArchiveFixtureEntry[]) => zlib.gzipSync(createTarArchive(entries)), + }, + { + label: ".zip", + path: "fixture-subpath.zip", + create: (entries: ArchiveFixtureEntry[]) => createZipArchive(entries), + }, + ]) { + it(`should read ${archiveCase.label} subpaths`, async () => { + const archivePath = path.join(testDir, archiveCase.path); + fs.writeFileSync( + archivePath, + archiveCase.create([ + { path: "pkg/README.md", content: "# Archive README\nLine 2\nLine 3\n" }, + { path: "pkg/src/index.ts", content: "export const archiveValue = 4;\n" }, + ]), + ); + + const result = await readTool.execute("test-call-archive-subpath", { + path: `${archivePath}:pkg/README.md`, + limit: 2, + }); + const output = getTextOutput(result); + + expect(output).toContain("# Archive README"); + expect(output).toContain("Line 2"); + expect(output).not.toContain("Line 3"); + expect(output).toContain("Use offset=3"); + }); + } + it("should detect image MIME type from file magic (not extension)", async () => { const png1x1Base64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+X2Z0AAAAASUVORK5CYII=";