refactor(coding-agent): split read tool into per-source modules

- ReadTool mixed plain-file reading with archive, sqlite, pdf-image, summary,
  selector, formatting and renderer concerns in one 3763-line module.
- Each now owns a sibling module; read.ts drops to 2020 lines and keeps its
  public exports, including the readToolRenderer re-export required because
  tools/index.ts star-exports ./read through the explicit ./tools entry.
- The pdfImageExtractions map and summaryParseCaches WeakMap stay single
  instances; execute() was deliberately left intact.
This commit is contained in:
can1357
2026-08-08 06:32:01 +02:00
parent 8cf3dce6fb
commit 7454b6e78f
9 changed files with 1952 additions and 1820 deletions
@@ -0,0 +1,209 @@
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import type { TextContent } from "@oh-my-pi/pi-ai";
import type { ToolSession } from "../sdk";
import { truncateHead } from "../session/streaming-output";
import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "../utils/zip";
import { applyListLimit } from "./list-limit";
import { resolveReadPath } from "./path-utils";
import type { ReadToolDetails } from "./read";
import {
buildInMemoryMultiRangeResult,
buildInMemoryTextResult,
decodeUtf8Text,
markMarkdownContentType,
prependSuffixResolutionNotice,
} from "./read-format";
import {
findSuffixMatchCached,
isNotFoundError,
isRemoteMountPath,
type SuffixMatchCache,
} from "./read-path-resolution";
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
import { formatBytes } from "./render-utils";
import { ToolError, throwIfAborted } from "./tool-errors";
import { toolResult } from "./tool-result";
interface ResolvedArchiveReadPath {
absolutePath: string;
archiveSubPath: string;
suffixResolution?: { from: string; to: string };
}
export async function resolveArchiveReadPath(
session: ToolSession,
readPath: string,
suffixCache: SuffixMatchCache,
signal?: AbortSignal,
): Promise<ResolvedArchiveReadPath | null> {
const candidates = parseArchivePathCandidates(readPath);
for (const candidate of candidates) {
let absolutePath = resolveReadPath(candidate.archivePath, session.cwd);
let suffixResolution: { from: string; to: string } | undefined;
try {
const stat = await Bun.file(absolutePath).stat();
if (stat.isDirectory()) continue;
return {
absolutePath,
archiveSubPath: candidate.archivePath === readPath ? "" : candidate.subPath,
suffixResolution,
};
} catch (error) {
if (!isNotFoundError(error) || isRemoteMountPath(absolutePath)) continue;
const suffixMatch = await findSuffixMatchCached(session, suffixCache, candidate.archivePath, signal);
if (!suffixMatch) continue;
try {
const retryStat = await Bun.file(suffixMatch.absolutePath).stat();
if (retryStat.isDirectory()) continue;
absolutePath = suffixMatch.absolutePath;
suffixResolution = { from: candidate.archivePath, to: suffixMatch.displayPath };
return {
absolutePath,
archiveSubPath: candidate.archivePath === readPath ? "" : candidate.subPath,
suffixResolution,
};
} catch (retryError) {
if (!isNotFoundError(retryError)) {
throw retryError;
}
}
}
}
return null;
}
async function readArchiveDirectory(
archive: ArchiveReader,
archivePath: string,
subPath: string,
offset: number | undefined,
limit: number | undefined,
details: ReadToolDetails,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
const DEFAULT_LIMIT = 500;
const effectiveLimit = limit ?? DEFAULT_LIMIT;
const allEntries = archive.listDirectory(subPath);
// `offset` is 1-indexed (line-selector semantics): `a.zip:dir:50` starts
// the listing at the 50th entry instead of being silently ignored.
const entries = offset !== undefined && offset > 1 ? allEntries.slice(offset - 1) : allEntries;
const listLimit = applyListLimit(entries, { limit: effectiveLimit });
const limitedEntries = listLimit.items;
const limitMeta = listLimit.meta;
for (let index = 0; index < limitedEntries.length; index++) {
throwIfAborted(signal);
}
const results = formatArchiveEntryLines(limitedEntries);
const output = results.length > 0 ? results.join("\n") : "(empty archive directory)";
const text = prependSuffixResolutionNotice(output, details.suffixResolution);
const truncation = truncateHead(text, { maxLines: Number.MAX_SAFE_INTEGER });
const directoryDetails: ReadToolDetails = { ...details, isDirectory: true };
const resultBuilder = toolResult<ReadToolDetails>(directoryDetails).text(truncation.content);
resultBuilder.sourcePath(archivePath).limits({ resultLimit: limitMeta.resultLimit?.reached });
if (truncation.truncated) {
directoryDetails.truncation = truncation;
resultBuilder.truncation(truncation, { direction: "head" });
}
return resultBuilder.done();
}
export async function readArchive(
session: ToolSession,
readPath: string,
parsedSel: ParsedSelector,
resolvedArchivePath: ResolvedArchiveReadPath,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
throwIfAborted(signal);
const archive = await openArchive(resolvedArchivePath.absolutePath);
throwIfAborted(signal);
const details: ReadToolDetails = markMarkdownContentType(
session,
{
resolvedPath: resolvedArchivePath.absolutePath,
suffixResolution: resolvedArchivePath.suffixResolution,
},
resolvedArchivePath.archiveSubPath,
);
let archiveSubPath = resolvedArchivePath.archiveSubPath;
let sel = parsedSel;
let node = archive.getNode(archiveSubPath);
if (!node && archiveSubPath) {
// `archive.zip:500` / `archive.zip:raw`: the whole subPath is a
// selector on the archive root, not a member name. Member names take
// precedence (getNode above); fall back to root + selector.
const wholeSel = parseSel(archiveSubPath);
if (wholeSel.kind !== "none") {
node = archive.getNode("");
archiveSubPath = "";
sel = wholeSel;
}
}
if (!node) {
throw new ToolError(`Path '${readPath}' not found inside archive`);
}
if (node.isDirectory) {
if (isMultiRange(sel)) {
throw new ToolError("Multi-range line selectors are not supported for archive directory listings.");
}
const { offset, limit } = selToOffsetLimit(sel);
return readArchiveDirectory(
archive,
resolvedArchivePath.absolutePath,
archiveSubPath,
offset,
limit,
details,
signal,
);
}
const entry = await archive.readFile(archiveSubPath);
const text = decodeUtf8Text(entry.bytes);
if (text === null) {
return toolResult<ReadToolDetails>(details)
.text(
prependSuffixResolutionNotice(
`[Cannot read binary archive entry '${entry.path}' (${formatBytes(entry.size)})]`,
resolvedArchivePath.suffixResolution,
),
)
.sourcePath(resolvedArchivePath.absolutePath)
.done();
}
// Archive members are immutable: there is no edit path for bytes inside
// an archive, and a hashline tag keyed to the archive file would invite
// (and fail) edits while clobbering sibling members' snapshots.
const raw = isRawSelector(sel);
const result =
isMultiRange(sel) && sel.kind === "lines"
? buildInMemoryMultiRangeResult(session, text, sel.ranges, {
details,
sourcePath: resolvedArchivePath.absolutePath,
entityLabel: "archive entry",
raw,
immutable: true,
})
: buildInMemoryTextResult(session, text, selToOffsetLimit(sel).offset, selToOffsetLimit(sel).limit, {
details,
sourcePath: resolvedArchivePath.absolutePath,
entityLabel: "archive entry",
raw,
immutable: true,
});
const firstText = result.content.find((content): content is TextContent => content.type === "text");
if (firstText) {
firstText.text = prependSuffixResolutionNotice(firstText.text, resolvedArchivePath.suffixResolution);
}
return result;
}
@@ -0,0 +1,593 @@
import * as path from "node:path";
import { formatHashlineHeader, formatNumberedLine, formatNumberedLines } from "@oh-my-pi/hashline";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { canonicalSnapshotKey, getFileSnapshotStore, recordSeenLines } from "../edit/file-snapshot-store";
import { normalizeToLF } from "../edit/normalize";
import { isMarkdownPath } from "../modes/theme/theme";
import type { ToolSession } from "../sdk";
import {
DEFAULT_MAX_BYTES,
noTruncResult,
type TruncationResult,
truncateHead,
truncateHeadBytes,
} from "../session/streaming-output";
import { buildLineEntriesWithBlockContext, type LineEntry, lineEntriesToPlainText } from "../utils/block-context";
import { resolveFileDisplayMode } from "../utils/file-display-mode";
import { formatPathRelativeToCwd, type LineRange } from "./path-utils";
import type { ReadToolDetails } from "./read";
import { formatBytes, shortenPath } from "./render-utils";
import { ToolError } from "./tool-errors";
import { toolResult } from "./tool-result";
function prependLineNumbers(text: string, startNum: number): string {
const textLines = text.split("\n");
return textLines.map((line, i) => `${startNum + i}|${line}`).join("\n");
}
export interface HashlineHeaderContext {
header: string;
tag: string;
fullText?: string;
}
export function formatReadHashlineHeader(displayPath: string, tag: string): string {
// In-workspace reads collapse to the bare filename for brevity: the edit
// tool's snapshot-tag recovery rebinds a bare `[name#tag]` onto the in-tree
// file it uniquely names. Out-of-workspace reads can't lean on that —
// recovery refuses to redirect a write outside the cwd/sandbox
// (HashlineFilesystem.allowTagPathRecovery) — so an absolute displayPath
// must stay directly resolvable, otherwise the basename resolves against
// cwd, misses, and the edit fails with "File not found" (e.g. ~/.claude/*).
// `shortenPath` keeps `~/.claude/...` (round-trips through resolveToCwd's ~
// expansion) instead of leaking the full home path into the read output.
const anchor = path.isAbsolute(displayPath) ? shortenPath(displayPath) : path.basename(displayPath);
return formatHashlineHeader(anchor, tag);
}
function recordFullHashlineContext(
session: ToolSession,
absolutePath: string | undefined,
displayPath: string,
fullText: string,
): HashlineHeaderContext | undefined {
if (!absolutePath || !path.isAbsolute(absolutePath)) return undefined;
const normalized = normalizeToLF(fullText);
const tag = getFileSnapshotStore(session).record(canonicalSnapshotKey(absolutePath), normalized);
return {
header: formatReadHashlineHeader(displayPath, tag),
tag,
fullText: normalized,
};
}
export async function readHashlineHeaderContext(
session: ToolSession,
absolutePath: string,
cwd: string,
): Promise<HashlineHeaderContext> {
const fullText = await Bun.file(absolutePath).text();
const context = recordFullHashlineContext(
session,
absolutePath,
formatPathRelativeToCwd(absolutePath, cwd),
fullText,
);
if (!context) throw new ToolError(`Cannot record hashline snapshot for non-absolute path: ${absolutePath}`);
return context;
}
export function hashlineHeaderContext(displayPath: string, tag: string): HashlineHeaderContext {
return { header: formatReadHashlineHeader(displayPath, tag), tag };
}
export function prependHashlineHeader(text: string, context: HashlineHeaderContext | undefined): string {
return context ? `${context.header}\n${text}` : text;
}
export function formatTextWithMode(
text: string,
startNum: number,
shouldAddHashLines: boolean,
shouldAddLineNumbers: boolean,
): string {
if (shouldAddHashLines) return formatNumberedLines(text, startNum);
if (shouldAddLineNumbers) return prependLineNumbers(text, startNum);
return text;
}
export const BRACKET_CONTEXT_ELLIPSIS = "…";
function formatLineEntryWithMode(entry: LineEntry, shouldAddHashLines: boolean, shouldAddLineNumbers: boolean): string {
if (entry.kind === "ellipsis") return BRACKET_CONTEXT_ELLIPSIS;
return formatSingleLine(entry.lineNumber, entry.text, shouldAddHashLines, shouldAddLineNumbers);
}
export function formatLineEntriesWithMode(
entries: readonly LineEntry[],
shouldAddHashLines: boolean,
shouldAddLineNumbers: boolean,
): string {
return entries.map(entry => formatLineEntryWithMode(entry, shouldAddHashLines, shouldAddLineNumbers)).join("\n");
}
const BRACE_PAIRS: Record<string, string> = { "{": "}", "(": ")", "[": "]" };
const BRACE_TAIL_TRAILING_RE = /^[;,)\]}]*$/;
/**
* Decide whether the kept lines surrounding an elided range collapse to a
* single brace-pair line in the rendered summary. Returns true when the head
* line ends with `{` / `(` / `[` and the tail line is the matching closer
* (optionally followed by terminating punctuation like `;`, `,`, or further
* closers — e.g. `};`, `})`, `]);`).
*/
export function canMergeBracePair(headLine: string, tailLine: string): boolean {
const head = headLine.trimEnd();
const tail = tailLine.trim();
const opener = head.slice(-1);
const closer = BRACE_PAIRS[opener];
if (!closer) return false;
if (!tail.startsWith(closer)) return false;
return BRACE_TAIL_TRAILING_RE.test(tail.slice(closer.length));
}
export function formatSingleLine(
line: number,
text: string,
shouldAddHashLines: boolean,
shouldAddLineNumbers: boolean,
): string {
if (shouldAddHashLines) return formatNumberedLine(line, text);
if (shouldAddLineNumbers) return `${line}|${text}`;
return text;
}
export function formatMergedBraceLine(
startLine: number,
endLine: number,
headText: string,
tailText: string,
shouldAddHashLines: boolean,
shouldAddLineNumbers: boolean,
): { model: string; display: string } {
const merged = `${headText.trimEnd()} … ${tailText.trim()}`;
if (shouldAddHashLines) {
return { model: `${startLine}-${endLine}:${merged}`, display: merged };
}
if (shouldAddLineNumbers) {
return { model: `${startLine}-${endLine}|${merged}`, display: merged };
}
return { model: merged, display: merged };
}
export function countTextLines(text: string): number {
if (text.length === 0) return 0;
// Count newlines directly instead of allocating an array via split("\n").
// Called on every read of file content; the result is identical (N newlines
// ⇒ N+1 lines for non-empty text).
let lines = 1;
for (let i = 0; i < text.length; i++) {
if (text.charCodeAt(i) === 10) lines++;
}
return lines;
}
export function contiguousLineNumbers(startLine: number, count: number): number[] {
const lines: number[] = [];
for (let offset = 0; offset < count; offset++) lines.push(startLine + offset);
return lines;
}
export function lineNumbersFromSpans(spans: readonly { startLine: number; endLine: number }[]): number[] {
const lines: number[] = [];
for (const span of spans) {
for (let line = span.startLine; line <= span.endLine; line++) lines.push(line);
}
return lines;
}
function recordInMemorySeenLines(
session: ToolSession,
absolutePath: string | undefined,
fullText: string,
seenLines: readonly number[] | undefined,
): void {
if (!absolutePath || !path.isAbsolute(absolutePath) || !seenLines || seenLines.length === 0) return;
getFileSnapshotStore(session).record(canonicalSnapshotKey(absolutePath), normalizeToLF(fullText), seenLines);
}
function lineNumbersFromEntries(entries: readonly LineEntry[]): number[] {
const lines: number[] = [];
for (const entry of entries) {
if (entry.kind === "line") lines.push(entry.lineNumber);
}
return lines;
}
/** Inclusive line range describing one elided span in a structural summary. */
export interface ElidedRange {
start: number;
end: number;
}
/** Sample ranges shown in the footer to demonstrate the multi-range syntax. */
const FOOTER_RANGE_SAMPLES = 2;
/**
* Footer appended to summarized reads telling the model how to recover the
* elided body. Without this hint, agents either ignore the `…`/`{ … }`
* markers or burn a turn guessing the right selector (see issue #1046). The
* footer demonstrates the multi-range selector syntax with concrete sample
* ranges drawn from the actual elision so the model re-reads only what it
* needs instead of falling back to `:raw` or whole-file reads.
*/
export function formatSummaryElisionFooter(
readPath: string,
elidedRanges: ReadonlyArray<ElidedRange>,
elidedLines: number,
): string {
if (elidedRanges.length === 0) return "";
const sampleCount = Math.min(elidedRanges.length, FOOTER_RANGE_SAMPLES);
const selector = elidedRanges
.slice(0, sampleCount)
.map(r => `${r.start}-${r.end}`)
.join(",");
const example = `${readPath}:${selector}`;
const tail = elidedRanges.length > sampleCount ? `, e.g. ${example}` : ` with ${example}`;
return `[…${elidedLines}ln elided; re-read needed ranges${tail}]`;
}
export const READ_CHUNK_SIZE = 8 * 1024;
/**
* Context lines added around an explicit range read. Anchor-stale failures
* cluster on edits whose anchors land just outside the most recent read
* window, but the data (`scripts/session-stats/analyze_selector_reads.py`)
* shows most follow-up reads are disjoint hops, not adjacent extensions —
* so symmetric padding rarely pays for itself.
*
* Leading=1 catches accidental single-line reads where the anchor is the
* line immediately above the requested start. Trailing=3 buffers the
* common case where the agent asks for a narrow range and then needs the
* next few lines to disambiguate an anchor.
*/
export const RANGE_LEADING_CONTEXT_LINES = 1;
export const RANGE_TRAILING_CONTEXT_LINES = 3;
/**
* Expand a [start, end) range with leading/trailing context lines on the
* sides where the user actually constrained the range. A start of 0 (no
* explicit offset) does not get leading context — that's already an
* open-ended read from the top.
*/
function expandRangeWithContext(
requestedStart: number,
requestedEnd: number,
totalLines: number,
expandStart: boolean,
expandEnd: boolean,
): { startLine: number; endLine: number } {
return {
startLine: expandStart ? Math.max(0, requestedStart - RANGE_LEADING_CONTEXT_LINES) : requestedStart,
endLine: expandEnd ? Math.min(totalLines, requestedEnd + RANGE_TRAILING_CONTEXT_LINES) : requestedEnd,
};
}
export function buildInMemoryTextResult(
session: ToolSession,
text: string,
offset: number | undefined,
limit: number | undefined,
options: {
details?: ReadToolDetails;
sourcePath?: string;
sourceUrl?: string;
sourceInternal?: string;
entityLabel: string;
ignoreResultLimits?: boolean;
raw?: boolean;
immutable?: boolean;
},
): AgentToolResult<ReadToolDetails> {
const displayMode = resolveFileDisplayMode(session, { raw: options.raw, immutable: options.immutable });
const details = options.details ?? {};
const allLines = text.split("\n");
const totalLines = allLines.length;
details.totalLines = totalLines;
// User-requested 0-indexed range start. Lines BEFORE this are leading
// context (added below if offset is explicit).
const requestedStart = offset ? Math.max(0, offset - 1) : 0;
const ignoreResultLimits = options.ignoreResultLimits ?? false;
const requestedEnd = limit !== undefined ? Math.min(requestedStart + limit, allLines.length) : allLines.length;
// Expand only on sides the user actually constrained: leading context
// when offset>1, trailing context when a finite limit was set. Raw mode
// never expands — without line numbers the padding is indistinguishable
// from requested content, so `raw:31-31` must return line 31 and nothing
// else (verbatim-extraction contract).
const rawDisplay = options.raw === true;
const expanded = expandRangeWithContext(
requestedStart,
requestedEnd,
allLines.length,
!rawDisplay && offset !== undefined && offset > 1,
!rawDisplay && limit !== undefined,
);
const startLine = expanded.startLine;
const endLineExpanded = expanded.endLine;
const startLineDisplay = startLine + 1;
const resultBuilder = toolResult(details);
if (options.sourcePath) {
resultBuilder.sourcePath(options.sourcePath);
}
if (options.sourceUrl) {
resultBuilder.sourceUrl(options.sourceUrl);
}
if (options.sourceInternal) {
resultBuilder.sourceInternal(options.sourceInternal);
}
if (requestedStart >= allLines.length) {
const suggestion =
allLines.length === 0
? `The ${options.entityLabel} is empty.`
: `Use :1 to read from the start, or :${allLines.length} to read the last line.`;
return resultBuilder
.text(
`Line ${requestedStart + 1} is beyond end of ${options.entityLabel} (${allLines.length} lines total). ${suggestion}`,
)
.done();
}
const endLine = endLineExpanded;
const selectedContent = allLines.slice(startLine, endLine).join("\n");
const userLimitedLines = limit !== undefined ? endLine - startLine : undefined;
const truncation = ignoreResultLimits ? noTruncResult(selectedContent) : truncateHead(selectedContent);
const shouldAddHashLines = displayMode.hashLines;
const shouldAddLineNumbers = shouldAddHashLines ? false : displayMode.lineNumbers;
const hashContext =
shouldAddHashLines && options.sourcePath
? recordFullHashlineContext(
session,
options.sourcePath,
formatPathRelativeToCwd(options.sourcePath, session.cwd),
text,
)
: undefined;
let emittedHashlineHeader = false;
let seenLines: number[] | undefined;
let rawSeenLines: number[] | undefined;
const formatText = (content: string, startNum: number): string => {
const lineCount = countTextLines(content);
details.displayContent = {
text: content,
startLine: startNum,
lineNumbers: Array.from({ length: lineCount }, (_, i) => startNum + i),
};
if (shouldAddHashLines) seenLines = contiguousLineNumbers(startNum, lineCount);
const formatted = formatTextWithMode(content, startNum, shouldAddHashLines, shouldAddLineNumbers);
if (!hashContext || emittedHashlineHeader) return formatted;
emittedHashlineHeader = true;
return prependHashlineHeader(formatted, hashContext);
};
const formatLineEntries = (entries: readonly LineEntry[], startNum: number): string => {
const firstLine = entries.find(entry => entry.kind === "line");
details.displayContent = {
text: lineEntriesToPlainText(entries, BRACKET_CONTEXT_ELLIPSIS),
startLine: firstLine?.kind === "line" ? firstLine.lineNumber : startNum,
lineNumbers: entries.map(entry => (entry.kind === "line" ? entry.lineNumber : null)),
};
if (shouldAddHashLines) seenLines = lineNumbersFromEntries(entries);
const formatted = formatLineEntriesWithMode(entries, shouldAddHashLines, shouldAddLineNumbers);
if (!hashContext || emittedHashlineHeader) return formatted;
emittedHashlineHeader = true;
return prependHashlineHeader(formatted, hashContext);
};
const buildLineEntries = (endLineDisplay: number): LineEntry[] =>
buildLineEntriesWithBlockContext(allLines, [{ startLine: startLineDisplay, endLine: endLineDisplay }], {
path: options.sourcePath,
});
let outputText: string;
let truncationInfo:
| { result: TruncationResult; options: { direction: "head"; startLine?: number; totalFileLines?: number } }
| undefined;
if (truncation.firstLineExceedsLimit) {
const firstLine = allLines[startLine] ?? "";
const firstLineBytes = Buffer.byteLength(firstLine, "utf-8");
const snippet = truncateHeadBytes(firstLine, DEFAULT_MAX_BYTES);
if (shouldAddHashLines) {
outputText = `[Line ${startLineDisplay} is ${formatBytes(
firstLineBytes,
)}, exceeds ${formatBytes(DEFAULT_MAX_BYTES)} limit. Hashline output requires full lines; cannot emit an editable numbered preview for a truncated line.]`;
} else {
outputText = formatText(snippet.text, startLineDisplay);
}
if (snippet.text.length === 0) {
outputText = `[Line ${startLineDisplay} is ${formatBytes(
firstLineBytes,
)}, exceeds ${formatBytes(DEFAULT_MAX_BYTES)} limit. Unable to display a valid UTF-8 snippet.]`;
}
details.truncation = truncation;
truncationInfo = {
result: truncation,
options: { direction: "head", startLine: startLineDisplay, totalFileLines: totalLines },
};
} else if (truncation.truncated) {
const outputLines = truncation.outputLines ?? countTextLines(truncation.content);
const endLineDisplay = startLineDisplay + Math.max(0, outputLines - 1);
if (options.raw === true) {
rawSeenLines = contiguousLineNumbers(startLineDisplay, outputLines);
outputText = formatText(truncation.content, startLineDisplay);
} else {
outputText = formatLineEntries(buildLineEntries(endLineDisplay), startLineDisplay);
}
details.truncation = truncation;
truncationInfo = {
result: truncation,
options: { direction: "head", startLine: startLineDisplay, totalFileLines: totalLines },
};
} else if (userLimitedLines !== undefined && startLine + userLimitedLines < allLines.length) {
const remaining = allLines.length - (startLine + userLimitedLines);
const nextOffset = startLine + userLimitedLines + 1;
if (options.raw === true) {
rawSeenLines = contiguousLineNumbers(startLineDisplay, userLimitedLines);
outputText = formatText(selectedContent, startLineDisplay);
} else {
outputText = formatLineEntries(buildLineEntries(endLine), startLineDisplay);
}
outputText += `\n\n[${remaining} more lines in ${options.entityLabel}. Use :${nextOffset} to continue]`;
} else {
if (options.raw === true) {
rawSeenLines = contiguousLineNumbers(startLineDisplay, endLine - startLine);
outputText = formatText(truncation.content, startLineDisplay);
} else {
outputText = formatLineEntries(buildLineEntries(endLine), startLineDisplay);
}
}
if (hashContext?.tag && options.sourcePath && seenLines) {
recordSeenLines(session, options.sourcePath, hashContext.tag, seenLines);
}
if (options.raw === true && options.sourcePath && options.immutable !== true && rawSeenLines) {
recordInMemorySeenLines(session, options.sourcePath, text, rawSeenLines);
}
resultBuilder.text(outputText);
if (truncationInfo) {
resultBuilder.truncation(truncationInfo.result, truncationInfo.options);
}
return resultBuilder.done();
}
/**
* Render a multi-range read against in-memory text. Each range emits a
* formatted block with its own anchors / line numbers, blocks are joined
* with an elision separator, and ranges past EOF surface as `[…]` notices
* so the model can correct the next call. No leading/trailing context is
* added — multi-range callers always specify exact bounds.
*/
export function buildInMemoryMultiRangeResult(
session: ToolSession,
text: string,
ranges: readonly LineRange[],
options: {
details?: ReadToolDetails;
sourcePath?: string;
sourceUrl?: string;
sourceInternal?: string;
entityLabel: string;
raw?: boolean;
immutable?: boolean;
},
): AgentToolResult<ReadToolDetails> {
const displayMode = resolveFileDisplayMode(session, { raw: options.raw, immutable: options.immutable });
const details = options.details ?? {};
const allLines = text.split("\n");
const totalLines = allLines.length;
details.totalLines = totalLines;
const shouldAddHashLines = displayMode.hashLines;
const shouldAddLineNumbers = shouldAddHashLines ? false : displayMode.lineNumbers;
const hashContext =
shouldAddHashLines && options.sourcePath
? recordFullHashlineContext(
session,
options.sourcePath,
formatPathRelativeToCwd(options.sourcePath, session.cwd),
text,
)
: undefined;
let emittedHashlineHeader = false;
let seenLines: number[] | undefined;
const resultBuilder = toolResult(details);
if (options.sourcePath) resultBuilder.sourcePath(options.sourcePath);
if (options.sourceUrl) resultBuilder.sourceUrl(options.sourceUrl);
if (options.sourceInternal) resultBuilder.sourceInternal(options.sourceInternal);
const outOfBounds: LineRange[] = [];
const visibleSpans: Array<{ startLine: number; endLine: number }> = [];
const rawParts: string[] = [];
for (const range of ranges) {
if (range.startLine > totalLines) {
outOfBounds.push(range);
continue;
}
const effectiveEnd = Math.min(range.endLine ?? totalLines, totalLines);
visibleSpans.push({ startLine: range.startLine, endLine: effectiveEnd });
if (options.raw === true) {
rawParts.push(allLines.slice(range.startLine - 1, effectiveEnd).join("\n"));
}
}
let outputText = "";
if (options.raw === true) {
outputText = rawParts.length > 0 ? rawParts.join("\n\n…\n\n") : "";
} else if (visibleSpans.length > 0) {
const entries = buildLineEntriesWithBlockContext(allLines, visibleSpans, { path: options.sourcePath });
if (shouldAddHashLines) seenLines = lineNumbersFromEntries(entries);
const firstLine = entries.find(entry => entry.kind === "line");
if (firstLine?.kind === "line") {
details.displayContent = {
text: lineEntriesToPlainText(entries, BRACKET_CONTEXT_ELLIPSIS),
startLine: firstLine.lineNumber,
lineNumbers: entries.map(entry => (entry.kind === "line" ? entry.lineNumber : null)),
};
}
const formatted = formatLineEntriesWithMode(entries, shouldAddHashLines, shouldAddLineNumbers);
outputText = hashContext && !emittedHashlineHeader ? prependHashlineHeader(formatted, hashContext) : formatted;
if (hashContext) emittedHashlineHeader = true;
}
const notices: string[] = [];
for (const range of outOfBounds) {
const bound = range.endLine !== undefined ? `${range.startLine}-${range.endLine}` : `${range.startLine}`;
notices.push(`[Range ${bound} is beyond end of ${options.entityLabel} (${totalLines} lines total); skipped]`);
}
const finalText =
notices.length > 0 ? (outputText ? `${outputText}\n${notices.join("\n")}` : notices.join("\n")) : outputText;
if (hashContext?.tag && options.sourcePath && seenLines) {
recordSeenLines(session, options.sourcePath, hashContext.tag, seenLines);
}
if (options.raw === true && options.sourcePath && options.immutable !== true && visibleSpans.length > 0) {
recordInMemorySeenLines(session, options.sourcePath, text, lineNumbersFromSpans(visibleSpans));
}
resultBuilder.text(finalText);
return resultBuilder.done();
}
export function decodeUtf8Text(bytes: Uint8Array): string | null {
if (bytes.indexOf(0) !== -1) return null;
try {
return new TextDecoder("utf-8", { fatal: true }).decode(bytes);
} catch {
return null;
}
}
export function prependSuffixResolutionNotice(text: string, suffixResolution?: { from: string; to: string }): string {
if (!suffixResolution) return text;
const notice = `[Path '${suffixResolution.from}' not found; resolved to '${suffixResolution.to}' via suffix match]`;
return text ? `${notice}\n${text}` : notice;
}
/**
* Tag Markdown reads for the TUI's formatted preview, gated on the opt-in
* `read.renderMarkdown` setting. Off by default; when disabled, no local
* read is tagged `text/markdown`, so the renderer output is identical to
* the pre-setting behavior. Internal-URL reads keep their protocol-supplied
* `contentType` and render as Markdown regardless of the setting.
*/
export function markMarkdownContentType(
session: ToolSession,
details: ReadToolDetails,
filePath: string,
): ReadToolDetails {
if (!details.contentType && session.settings.get("read.renderMarkdown") && isMarkdownPath(filePath)) {
details.contentType = "text/markdown";
}
return details;
}
@@ -0,0 +1,36 @@
import * as path from "node:path";
import { getRemoteDir } from "@oh-my-pi/pi-utils";
import type { ToolSession } from "../sdk";
import { findUniqueWorkspaceSuffix } from "./path-utils";
// Remote mount path prefix (sshfs mounts) - skip fuzzy matching to avoid hangs
const REMOTE_MOUNT_PREFIX = getRemoteDir() + path.sep;
export function isRemoteMountPath(absolutePath: string): boolean {
return absolutePath.startsWith(REMOTE_MOUNT_PREFIX);
}
export function isNotFoundError(error: unknown): boolean {
if (!error || typeof error !== "object") return false;
const code = (error as { code?: string }).code;
return code === "ENOENT" || code === "ENOTDIR";
}
/** Per-execute memo of suffix-glob lookups; `null` records a confirmed miss. */
export type SuffixMatchCache = Map<string, { absolutePath: string; displayPath: string } | null>;
/**
* Memoized {@link findUniqueWorkspaceSuffix} for a single read call. A missing
* path with archive/sqlite extensions probes the workspace once per stage
* (archive candidates, sqlite candidates, plain path) — each glob carries a
* 5s timeout, so repeated lookups of the same string stack into a long
* stall before erroring. The cache collapses repeats within one execute().
*/
export async function findSuffixMatchCached(
session: ToolSession,
cache: SuffixMatchCache,
rawPath: string,
signal?: AbortSignal,
): Promise<{ absolutePath: string; displayPath: string } | null> {
const hit = cache.get(rawPath);
if (hit !== undefined) return hit;
const result = await findUniqueWorkspaceSuffix(rawPath, session.cwd, signal);
cache.set(rawPath, result);
return result;
}
@@ -0,0 +1,250 @@
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { isEexist, isEnotempty, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import type { ToolSession } from "../sdk";
import { loadImageInput, MAX_IMAGE_INPUT_BYTES, webpExclusionForModel } from "../utils/image-loading";
import { convertFileWithMarkit } from "../utils/markit";
import type { ReadToolDetails } from "./read";
import { prependSuffixResolutionNotice } from "./read-format";
import { isNotFoundError } from "./read-path-resolution";
import { formatBytes } from "./render-utils";
import { ToolError } from "./tool-errors";
import { toolResult } from "./tool-result";
const MAX_IMAGE_SIZE = MAX_IMAGE_INPUT_BYTES;
const PDF_IMAGE_PLACEHOLDER_RE = /<!--\s*image:\s*([^\s<>]+)(.*?)-->/g;
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
const PDF_IMAGE_MEMBER_EXTENSION_RE = /\.png$/i;
const PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH = 96;
interface PdfImageSnapshot {
directory: string;
filePath: string;
digest: string;
}
interface PdfImageExtraction {
controller: AbortController;
promise: Promise<string>;
settled: boolean;
waiters: number;
}
const pdfImageExtractions = new Map<string, PdfImageExtraction>();
function pdfImageMemberPath(pdfPath: string, imageId: string): string {
const member = PDF_IMAGE_MEMBER_EXTENSION_RE.test(imageId) ? imageId : `${imageId}.png`;
return `${pdfPath}:${member}`;
}
export function rewritePdfImagePlaceholders(markdown: string, pdfPath: string): string {
return markdown.replace(PDF_IMAGE_PLACEHOLDER_RE, (_match: string, imageId: string, metadataText: string) => {
const metadata = metadataText.trim();
const suffix = metadata.length > 0 ? ` (${metadata})` : "";
return `Image ${imageId}${suffix}: read \`${pdfImageMemberPath(pdfPath, imageId)}\``;
});
}
export function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; member: string } | null {
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
if (!match) return null;
const pdfPath = match[1];
const member = match[2];
if (pdfPath === undefined || member === undefined) return null;
if (member.length !== 0 && !PDF_IMAGE_MEMBER_EXTENSION_RE.test(member)) return null;
return { pdfPath, member };
}
function pdfImageCacheDir(session: ToolSession, absolutePdfPath: string, contentDigest: string): string {
const artifactsDir = session.getArtifactsDir?.();
let root = artifactsDir ?? undefined;
if (root === undefined) {
const sessionFile = session.getSessionFile();
root = sessionFile?.endsWith(".jsonl") ? sessionFile.slice(0, -6) : path.join(os.tmpdir(), "omp-read-pdf-images");
}
const basename = path
.basename(absolutePdfPath)
.replace(/[^A-Za-z0-9._-]/g, "_")
.slice(0, PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH);
const pathDigest = Bun.hash(absolutePdfPath).toString(36);
return path.join(root, "read-pdf-images", `${basename}-${pathDigest}-${contentDigest}`);
}
async function snapshotPdfSource(absolutePdfPath: string, signal?: AbortSignal): Promise<PdfImageSnapshot> {
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-read-pdf-"));
try {
const bytes = await untilAborted(signal, () => Bun.file(absolutePdfPath).bytes());
signal?.throwIfAborted();
const digest = new Bun.CryptoHasher("sha256").update(bytes).digest("hex");
const filePath = path.join(directory, "source.pdf");
await Bun.write(filePath, bytes);
signal?.throwIfAborted();
return { directory, filePath, digest };
} catch (error) {
await fs.rm(directory, { recursive: true, force: true });
throw error;
}
}
async function listPdfImageMembers(imageDir: string): Promise<string[]> {
try {
const entries = await fs.readdir(imageDir, { withFileTypes: true });
const members: string[] = [];
for (const entry of entries) {
if (entry.isFile() && PDF_IMAGE_MEMBER_EXTENSION_RE.test(entry.name)) members.push(entry.name);
}
return members.sort();
} catch (error) {
if (isNotFoundError(error)) return [];
throw error;
}
}
async function extractPdfImages(snapshot: PdfImageSnapshot, imageDir: string, signal: AbortSignal): Promise<string> {
const markerPath = path.join(imageDir, ".extracted");
try {
await fs.stat(markerPath);
return imageDir;
} catch (error) {
if (!isNotFoundError(error)) throw error;
}
await fs.mkdir(path.dirname(imageDir), { recursive: true });
const stagingDir = await fs.mkdtemp(`${imageDir}.tmp-`);
let published = false;
try {
const result = await convertFileWithMarkit(snapshot.filePath, signal, { imageDir: stagingDir });
if (!result.ok) {
throw new ToolError(`Cannot extract images from PDF: ${result.error ?? "conversion failed"}`);
}
await Bun.write(path.join(stagingDir, ".extracted"), "ok");
try {
await fs.rename(stagingDir, imageDir);
published = true;
} catch (error) {
if (!isEexist(error) && !isEnotempty(error)) throw error;
try {
await fs.stat(markerPath);
} catch (markerError) {
if (isNotFoundError(markerError)) throw error;
throw markerError;
}
}
return imageDir;
} finally {
if (!published) await fs.rm(stagingDir, { recursive: true, force: true });
}
}
function createPdfImageExtraction(snapshot: PdfImageSnapshot, imageDir: string): PdfImageExtraction {
const controller = new AbortController();
const promise = extractPdfImages(snapshot, imageDir, controller.signal).finally(() =>
fs.rm(snapshot.directory, { recursive: true, force: true }),
);
const extraction: PdfImageExtraction = { controller, promise, settled: false, waiters: 0 };
const settle = () => {
extraction.settled = true;
if (pdfImageExtractions.get(imageDir) === extraction) pdfImageExtractions.delete(imageDir);
};
void promise.then(settle, settle);
return extraction;
}
async function waitForPdfImageExtraction(
extraction: PdfImageExtraction,
signal: AbortSignal | undefined,
): Promise<string> {
extraction.waiters++;
try {
return await untilAborted(signal, extraction.promise);
} finally {
extraction.waiters--;
if (extraction.waiters === 0 && !extraction.settled) {
extraction.controller.abort();
try {
await extraction.promise;
} catch {}
}
}
}
async function ensurePdfImageCache(
session: ToolSession,
absolutePdfPath: string,
signal?: AbortSignal,
): Promise<string> {
const snapshot = await snapshotPdfSource(absolutePdfPath, signal);
const imageDir = pdfImageCacheDir(session, absolutePdfPath, snapshot.digest);
const existing = pdfImageExtractions.get(imageDir);
if (existing && !existing.settled && !existing.controller.signal.aborted) {
await fs.rm(snapshot.directory, { recursive: true, force: true });
return waitForPdfImageExtraction(existing, signal);
}
const extraction = createPdfImageExtraction(snapshot, imageDir);
pdfImageExtractions.set(imageDir, extraction);
return waitForPdfImageExtraction(extraction, signal);
}
export async function readPdfImageMember(
session: ToolSession,
autoResizeImages: boolean,
absolutePdfPath: string,
pdfDisplayPath: string,
member: string,
suffixResolution: { from: string; to: string } | undefined,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
const imageDir = await ensurePdfImageCache(session, absolutePdfPath, signal);
const members = await listPdfImageMembers(imageDir);
if (member.length === 0) {
const text =
members.length === 0
? "No extractable PDF image members found."
: `Extractable PDF image members:\n${members
.map(imageMember => `- read \`${pdfDisplayPath}:${imageMember}\``)
.join("\n")}`;
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.text(prependSuffixResolutionNotice(text, suffixResolution))
.sourcePath(absolutePdfPath)
.done();
}
if (!members.includes(member)) {
const available = members.length === 0 ? "(none)" : members.join(", ");
throw new ToolError(`PDF image member '${member}' not found. Available members: ${available}`);
}
const imagePath = path.join(imageDir, member);
const imageStat = await Bun.file(imagePath).stat();
if (imageStat.size > MAX_IMAGE_SIZE) {
const sizeStr = formatBytes(imageStat.size);
const maxStr = formatBytes(MAX_IMAGE_SIZE);
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
}
const metadata = await readImageMetadata(imagePath);
const mimeType = metadata?.mimeType;
if (!mimeType) throw new ToolError(`PDF image member '${member}' is not a supported image.`);
const imageInput = await loadImageInput({
path: `${pdfDisplayPath}:${member}`,
cwd: session.cwd,
autoResize: autoResizeImages,
maxBytes: MAX_IMAGE_SIZE,
resolvedPath: imagePath,
detectedMimeType: mimeType,
excludeWebP: webpExclusionForModel(session.getActiveModel?.()),
});
if (!imageInput) {
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
}
const textNote = prependSuffixResolutionNotice(imageInput.textNote, suffixResolution);
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.content([
{ type: "text", text: textNote },
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
])
.sourcePath(imageInput.resolvedPath)
.done();
}
@@ -0,0 +1,289 @@
import * as path from "node:path";
import type { Component } from "@oh-my-pi/pi-tui";
import { Text } from "@oh-my-pi/pi-tui";
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
import { fileHyperlink, renderCodeCell, renderMarkdownCell, renderStatusLine, tryResolveInternalUrlSync } from "../tui";
import { CachedOutputBlock, markFramedBlockComponent } from "../tui/output-block";
import { type ReadUrlToolDetails, renderReadUrlCall, renderReadUrlResult } from "./fetch";
import { formatFullOutputReference, formatStyledTruncationWarning, stripOutputNotice } from "./output-meta";
import { isReadableUrlPath, splitInternalUrlSel, splitPathAndSel } from "./path-utils";
import type { ReadToolDetails } from "./read";
import { isRawSelector, parseSel } from "./read-selector";
import { formatBytes, replaceTabs, shortenPath, wrapBrackets } from "./render-utils";
// =============================================================================
// TUI Renderer
// =============================================================================
interface ReadRenderArgs {
path?: unknown;
file_path?: unknown;
// Legacy fields from old schema — tolerated for in-flight tool calls during transition
offset?: number;
limit?: number;
raw?: boolean;
}
const INTERNAL_URL_LIKE_RE = /^[a-z][a-z0-9+.-]*:\/\//i;
function splitReadRenderPath(rawPath: string): { path: string; sel?: string } {
if (INTERNAL_URL_LIKE_RE.test(rawPath)) {
const internal = splitInternalUrlSel(rawPath);
if (internal.sel) return internal;
}
return splitPathAndSel(rawPath);
}
function firstReadSelectorLine(sel: string | undefined): number | undefined {
if (!sel) return undefined;
try {
const parsed = parseSel(sel);
if (parsed.kind !== "lines") return undefined;
return parsed.ranges[0].startLine;
} catch {
return undefined;
}
}
/** Absolute fs path the read result actually resolved to, used as the OSC 8 link
* target when the structured `resolvedPath` isn't set (the common plain-file and
* image reads only record the path in `meta.source`). URL/internal sources are
* not fs paths, so only `type: "path"` qualifies. */
function readSourceFsPath(details: ReadToolDetails | undefined): string | undefined {
const source = details?.meta?.source;
return source?.type === "path" ? source.value : undefined;
}
function formatReadPathLink(
rawPath: string,
options: {
resolvedPath?: string;
sourcePath?: string;
suffixResolution?: { from: string; to: string };
offset?: number;
fallbackLabel?: string;
},
): string {
const split = splitReadRenderPath(rawPath);
const basePath = split.path || rawPath;
const selectorSuffix = split.sel ? `:${split.sel}` : "";
const plainDisplayPath = options.suffixResolution
? shortenPath(options.suffixResolution.to)
: shortenPath(basePath || options.resolvedPath || options.fallbackLabel || rawPath);
const absoluteInputPath = path.isAbsolute(basePath) ? basePath : undefined;
const target =
options.resolvedPath ?? options.sourcePath ?? tryResolveInternalUrlSync(basePath) ?? absoluteInputPath;
const line = firstReadSelectorLine(split.sel) ?? options.offset;
const linkOptions = line !== undefined ? { line } : undefined;
const linkedPath = target ? fileHyperlink(target, plainDisplayPath, linkOptions) : plainDisplayPath;
return `${linkedPath}${selectorSuffix}`;
}
export const readToolRenderer = {
renderCall(args: ReadRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {
const rawPath =
typeof args.file_path === "string" ? args.file_path : typeof args.path === "string" ? args.path : "";
if (isReadableUrlPath(rawPath)) {
return renderReadUrlCall({ path: rawPath, raw: args.raw }, _options, uiTheme);
}
const offset = args.offset;
const limit = args.limit;
let pathDisplay = formatReadPathLink(rawPath, { offset, fallbackLabel: "…" }) || "…";
if (offset !== undefined || limit !== undefined) {
const startLine = offset ?? 1;
const endLine = limit !== undefined ? startLine + limit - 1 : "";
pathDisplay += `:${startLine}${endLine ? `-${endLine}` : ""}`;
}
const text = renderStatusLine({ icon: "pending", title: "Read", description: pathDisplay }, uiTheme);
return new Text(text, 0, 0);
},
renderResult(
result: { content: Array<{ type: string; text?: string }>; details?: ReadToolDetails; isError?: boolean },
options: RenderResultOptions,
uiTheme: Theme,
args?: ReadRenderArgs,
): Component {
const urlDetails = result.details as ReadUrlToolDetails | undefined;
const baseRawPathForKind =
typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : "";
if (urlDetails?.kind === "url" || isReadableUrlPath(baseRawPathForKind)) {
return renderReadUrlResult(
result as {
content: Array<{ type: string; text?: string }>;
details?: ReadUrlToolDetails;
isError?: boolean;
},
options,
uiTheme,
);
}
if (result.isError) {
const rawErrorText = result.content?.find(c => c.type === "text")?.text ?? "";
const errorText = (rawErrorText || "Unknown error").replace(/^Error:\s*/, "");
const rawPath =
typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : "";
const filePath =
formatReadPathLink(rawPath, { offset: args?.offset, sourcePath: readSourceFsPath(result.details) }) ||
shortenPath(rawPath);
let title = filePath ? `Read ${filePath}` : "Read";
if (args?.offset !== undefined || args?.limit !== undefined) {
const startLine = args.offset ?? 1;
const endLine = args.limit !== undefined ? startLine + args.limit - 1 : "";
title += `:${startLine}${endLine ? `-${endLine}` : ""}`;
}
const header = renderStatusLine({ icon: "error", title }, uiTheme);
const errorLines = errorText.split("\n").map(line => uiTheme.fg("error", replaceTabs(line)));
const outputBlock = new CachedOutputBlock();
return markFramedBlockComponent({
render: (width: number) =>
outputBlock.render({ header, state: "error", sections: [{ lines: errorLines }], width }, uiTheme),
invalidate: () => outputBlock.invalidate(),
});
}
const details = result.details;
const rawText = result.content?.find(c => c.type === "text")?.text ?? "";
// Prefer structured `displayContent` from details when available so the TUI
// shows clean file content (no model-only hashline anchors) without parsing the formatted text.
// Fall back to the raw text, but strip the LLM-facing notice so it doesn't
// echo next to the styled warning line below.
const contentText = details?.displayContent?.text ?? stripOutputNotice(rawText, details?.meta);
const imageContent = result.content?.find(c => c.type === "image");
const rawPath =
typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : "";
const renderPath = splitReadRenderPath(rawPath);
const lang = getLanguageFromPath(renderPath.path);
const warningLines: string[] = [];
const truncation = details?.meta?.truncation;
const fallback = details?.truncation;
if (details?.resolvedPath) {
warningLines.push(uiTheme.fg("dim", wrapBrackets(`Resolved path: ${details.resolvedPath}`, uiTheme)));
}
if (truncation) {
if (fallback?.firstLineExceedsLimit) {
let warning = `First line exceeds ${formatBytes(fallback.outputBytes ?? fallback.totalBytes)} limit`;
if (truncation.artifactId) {
warning += `. ${formatFullOutputReference(truncation.artifactId)}`;
}
warningLines.push(uiTheme.fg("warning", wrapBrackets(warning, uiTheme)));
} else {
const warning = formatStyledTruncationWarning(details?.meta, uiTheme);
if (warning) warningLines.push(warning);
}
}
if (imageContent) {
const suffix = details?.suffixResolution;
const displayPath = formatReadPathLink(rawPath, {
resolvedPath: details?.resolvedPath,
sourcePath: readSourceFsPath(details),
suffixResolution: suffix,
fallbackLabel: "image",
});
const correction = suffix ? ` ${uiTheme.fg("dim", `(corrected from ${shortenPath(suffix.from)})`)}` : "";
const header = renderStatusLine(
{ icon: suffix ? "warning" : "success", title: "Read", description: `${displayPath}${correction}` },
uiTheme,
);
const detailLines = contentText ? contentText.split("\n").map(line => uiTheme.fg("toolOutput", line)) : [];
const lines = [...detailLines, ...warningLines];
const outputBlock = new CachedOutputBlock();
return markFramedBlockComponent({
render: (width: number) =>
outputBlock.render(
{
header,
state: "success",
sections: [
{
label: uiTheme.fg("toolTitle", "Details"),
lines: lines.length > 0 ? lines : [uiTheme.fg("dim", "(image)")],
},
],
width,
},
uiTheme,
),
invalidate: () => outputBlock.invalidate(),
});
}
const suffix = details?.suffixResolution;
// resolvedPath is the absolute fs path when a read resolved/corrected the
// input (suffix match, internal URL, archive/sqlite/notebook); plain file
// reads only record the absolute path in meta.source, so fall back to that
// (and then to a sync internal-URL resolver) to keep the title clickable.
const displayPath = formatReadPathLink(rawPath, {
resolvedPath: details?.resolvedPath,
sourcePath: readSourceFsPath(details),
suffixResolution: suffix,
offset: args?.offset,
});
const correction = suffix ? ` ${uiTheme.fg("dim", `(corrected from ${shortenPath(suffix.from)})`)}` : "";
let title = displayPath ? `Read ${displayPath}${correction}` : "Read";
if (args?.offset !== undefined || args?.limit !== undefined) {
const startLine = args.offset ?? 1;
const endLine = args.limit !== undefined ? startLine + args.limit - 1 : "";
title += `:${startLine}${endLine ? `-${endLine}` : ""}`;
}
if (details?.summary) {
title += ` (summary: ${details.summary.elidedSpans} elided span${details.summary.elidedSpans === 1 ? "" : "s"})`;
}
if (details?.conflictCount && details.conflictCount > 0) {
const n = details.conflictCount;
title += ` ${uiTheme.fg("warning", `(⚠ ${n} conflict${n === 1 ? "" : "s"})`)}`;
}
const rawRequested = args?.raw === true || isRawSelector(parseSel(renderPath.sel));
const isMarkdown = details?.contentType === "text/markdown" && !rawRequested;
let cachedWidth: number | undefined;
let cachedExpanded: boolean | undefined;
let cachedLines: string[] | undefined;
return markFramedBlockComponent({
render: (width: number) => {
const expanded = options.expanded;
if (cachedLines && cachedWidth === width && cachedExpanded === expanded) return cachedLines;
cachedLines = isMarkdown
? renderMarkdownCell(
{
content: contentText,
title,
status: "complete",
output: warningLines.length > 0 ? warningLines.join("\n") : undefined,
expanded,
width,
},
uiTheme,
)
: renderCodeCell(
{
code: contentText,
language: lang,
title,
status: "complete",
output: warningLines.length > 0 ? warningLines.join("\n") : undefined,
expanded,
codeStartLine: details?.displayContent?.startLine,
codeLineNumbers: details?.displayContent?.lineNumbers,
width,
},
uiTheme,
);
cachedWidth = width;
cachedExpanded = expanded;
return cachedLines;
},
invalidate: () => {
cachedWidth = undefined;
cachedExpanded = undefined;
cachedLines = undefined;
},
});
},
mergeCallAndResult: true,
};
@@ -0,0 +1,84 @@
import type { LineRange } from "./path-utils";
import { parseLineRanges } from "./path-utils";
import { ToolError } from "./tool-errors";
/** Parsed representation of a path-embedded selector. */
export type ParsedSelector =
| { kind: "none" }
| { kind: "raw" }
| { kind: "conflicts" }
| { kind: "lines"; ranges: [LineRange, ...LineRange[]]; raw?: boolean };
/** Returns true when the selector requested verbatim/raw output (alone or combined with a range). */
export function isRawSelector(parsed: ParsedSelector): boolean {
return parsed.kind === "raw" || (parsed.kind === "lines" && parsed.raw === true);
}
/** Returns true when the selector requested multiple line ranges. */
export function isMultiRange(parsed: ParsedSelector): boolean {
return parsed.kind === "lines" && parsed.ranges.length > 1;
}
function selectorChunkLooksReadLike(chunk: string): boolean {
const lower = chunk.toLowerCase();
return (
lower === "raw" || lower === "conflicts" || /^-\d+(?:[-+]\d+)?$/.test(chunk) || parseLineRanges(chunk) !== null
);
}
function invalidSelector(sel: string): ToolError {
return new ToolError(
`Invalid selector ':${sel}'. Use :N, :N-M, :N+K, :N- (open-ended), a comma-separated list of ranges, :raw, or a range combined with raw (e.g. :raw:50-100).`,
);
}
export function parseSel(sel: string | undefined): ParsedSelector {
if (!sel || sel.length === 0) return { kind: "none" };
// Compound selector: `1-50:raw` or `raw:1-50`. Split into chunks and accept
// exactly one line range (possibly multi) plus the literal `raw`. Selector-like
// compounds that are not in that accepted set are invalid rather than "none";
// otherwise `read` can silently widen a malformed selector like
// `artifact://5:conflicts:1-1` while `grep` rejects it.
if (sel.includes(":")) {
const chunks = sel.split(":");
if (chunks.length === 2) {
const [a, b] = chunks as [string, string];
const aIsRaw = a.toLowerCase() === "raw";
const bIsRaw = b.toLowerCase() === "raw";
const rangeChunk = aIsRaw ? b : bIsRaw ? a : null;
const rawChunk = aIsRaw ? a : bIsRaw ? b : null;
if (rangeChunk !== null && rawChunk !== null) {
const ranges = parseLineRanges(rangeChunk);
if (ranges) {
return { kind: "lines", ranges, raw: true };
}
}
}
if (chunks.every(selectorChunkLooksReadLike)) throw invalidSelector(sel);
// Unrecognized compound — fall through (sqlite/archive/url consume their own colon syntax).
return { kind: "none" };
}
if (sel.toLowerCase() === "raw") return { kind: "raw" };
if (sel.toLowerCase() === "conflicts") return { kind: "conflicts" };
const ranges = parseLineRanges(sel);
if (ranges) {
return { kind: "lines", ranges };
}
// Unrecognized selectors fall through; sqlite/archive/url readers consume their own colon syntax.
return { kind: "none" };
}
/**
* Convert a single-range selector to the offset/limit pair used by internal pagination.
* Returns the FIRST range only — multi-range callers MUST branch on `isMultiRange` before
* calling this helper.
*/
export function selToOffsetLimit(parsed: ParsedSelector): { offset?: number; limit?: number } {
if (parsed.kind === "lines") {
const first = parsed.ranges[0];
const limit = first.endLine !== undefined ? first.endLine - first.startLine + 1 : undefined;
return { offset: first.startLine, limit };
}
return {};
}
@@ -0,0 +1,215 @@
import { Database } from "bun:sqlite";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import type { ToolSession } from "../sdk";
import { DEFAULT_MAX_LINES, truncateHead } from "../session/streaming-output";
import { applyListLimit } from "./list-limit";
import { resolveReadPath } from "./path-utils";
import type { ReadToolDetails } from "./read";
import { prependSuffixResolutionNotice } from "./read-format";
import {
findSuffixMatchCached,
isNotFoundError,
isRemoteMountPath,
type SuffixMatchCache,
} from "./read-path-resolution";
import {
executeReadQuery,
getRowByKey,
getRowByRowId,
getTableSchema,
isSqliteFile,
listTables,
MAX_RAW_QUERY_ROWS,
parseSqlitePathCandidates,
parseSqliteSelector,
queryRows,
renderRow,
renderSchema,
renderTable,
renderTableList,
resolveTableRowLookup,
} from "./sqlite-reader";
import { ToolError, throwIfAborted } from "./tool-errors";
import { toolResult } from "./tool-result";
interface ResolvedSqliteReadPath {
absolutePath: string;
sqliteSubPath: string;
queryString: string;
suffixResolution?: { from: string; to: string };
}
export async function resolveSqliteReadPath(
session: ToolSession,
readPath: string,
suffixCache: SuffixMatchCache,
signal?: AbortSignal,
): Promise<ResolvedSqliteReadPath | null> {
const candidates = parseSqlitePathCandidates(readPath);
for (const candidate of candidates) {
let absolutePath = resolveReadPath(candidate.sqlitePath, session.cwd);
let suffixResolution: { from: string; to: string } | undefined;
try {
const stat = await Bun.file(absolutePath).stat();
if (stat.isDirectory()) continue;
if (!(await isSqliteFile(absolutePath))) continue;
return {
absolutePath,
sqliteSubPath: candidate.subPath,
queryString: candidate.queryString,
suffixResolution,
};
} catch (error) {
if (!isNotFoundError(error) || isRemoteMountPath(absolutePath)) continue;
const suffixMatch = await findSuffixMatchCached(session, suffixCache, candidate.sqlitePath, signal);
if (!suffixMatch) continue;
try {
const retryStat = await Bun.file(suffixMatch.absolutePath).stat();
if (retryStat.isDirectory()) continue;
if (!(await isSqliteFile(suffixMatch.absolutePath))) continue;
absolutePath = suffixMatch.absolutePath;
suffixResolution = { from: candidate.sqlitePath, to: suffixMatch.displayPath };
return {
absolutePath,
sqliteSubPath: candidate.subPath,
queryString: candidate.queryString,
suffixResolution,
};
} catch (retryError) {
if (!isNotFoundError(retryError)) {
throw retryError;
}
}
}
}
return null;
}
export async function readSqlite(
resolvedSqlitePath: ResolvedSqliteReadPath,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
throwIfAborted(signal);
const selectorInput = {
subPath: resolvedSqlitePath.sqliteSubPath,
queryString: resolvedSqlitePath.queryString,
};
const selector = parseSqliteSelector(selectorInput.subPath, selectorInput.queryString);
const details: ReadToolDetails = {
resolvedPath: resolvedSqlitePath.absolutePath,
suffixResolution: resolvedSqlitePath.suffixResolution,
};
let db: Database | null = null;
try {
db = new Database(resolvedSqlitePath.absolutePath, { readonly: true, strict: true });
db.run("PRAGMA busy_timeout = 3000");
throwIfAborted(signal);
switch (selector.kind) {
case "list": {
const listLimit = applyListLimit(listTables(db), { limit: 500 });
const output = prependSuffixResolutionNotice(
renderTableList(listLimit.items),
resolvedSqlitePath.suffixResolution,
);
const truncation = truncateHead(output, { maxLines: Number.MAX_SAFE_INTEGER });
details.truncation = truncation.truncated ? truncation : undefined;
const resultBuilder = toolResult<ReadToolDetails>(details)
.text(truncation.content)
.sourcePath(resolvedSqlitePath.absolutePath)
.limits({ resultLimit: listLimit.meta.resultLimit?.reached });
if (truncation.truncated) {
resultBuilder.truncation(truncation, { direction: "head" });
}
return resultBuilder.done();
}
case "schema": {
const sampleRows = queryRows(db, selector.table, { limit: selector.sampleLimit, offset: 0 });
let output = renderSchema(getTableSchema(db, selector.table), {
columns: sampleRows.columns,
rows: sampleRows.rows,
});
if (sampleRows.rows.length < sampleRows.totalCount) {
const remaining = sampleRows.totalCount - sampleRows.rows.length;
output += `\n[${remaining} more rows; append :${selector.table}?limit=20&offset=${sampleRows.rows.length} to the database path to continue]`;
}
return toolResult<ReadToolDetails>(details)
.text(prependSuffixResolutionNotice(output, resolvedSqlitePath.suffixResolution))
.sourcePath(resolvedSqlitePath.absolutePath)
.done();
}
case "row": {
const lookup = resolveTableRowLookup(db, selector.table);
const row =
lookup.kind === "pk"
? getRowByKey(db, selector.table, lookup, selector.key)
: getRowByRowId(db, selector.table, selector.key);
if (!row) {
return toolResult<ReadToolDetails>(details)
.text(
prependSuffixResolutionNotice(
`No row found in table '${selector.table}' for key '${selector.key}'.`,
resolvedSqlitePath.suffixResolution,
),
)
.sourcePath(resolvedSqlitePath.absolutePath)
.done();
}
return toolResult<ReadToolDetails>(details)
.text(prependSuffixResolutionNotice(renderRow(row), resolvedSqlitePath.suffixResolution))
.sourcePath(resolvedSqlitePath.absolutePath)
.done();
}
case "query": {
const page = queryRows(db, selector.table, selector);
return toolResult<ReadToolDetails>(details)
.text(
prependSuffixResolutionNotice(
renderTable(page.columns, page.rows, {
totalCount: page.totalCount,
offset: selector.offset,
limit: selector.limit,
table: selector.table,
dbPath: resolvedSqlitePath.absolutePath,
}),
resolvedSqlitePath.suffixResolution,
),
)
.sourcePath(resolvedSqlitePath.absolutePath)
.done();
}
case "raw": {
const result = executeReadQuery(db, selector.sql);
let output = renderTable(result.columns, result.rows, {
totalCount: result.rows.length,
offset: 0,
limit: result.rows.length || DEFAULT_MAX_LINES,
table: "query",
dbPath: resolvedSqlitePath.absolutePath,
});
if (result.truncated) {
output += `\n[Output capped at ${MAX_RAW_QUERY_ROWS} rows; add a LIMIT/OFFSET clause to the query to page through more]`;
}
return toolResult<ReadToolDetails>(details)
.text(prependSuffixResolutionNotice(output, resolvedSqlitePath.suffixResolution))
.sourcePath(resolvedSqlitePath.absolutePath)
.done();
}
}
throw new ToolError("Unsupported SQLite selector");
} catch (error) {
if (error instanceof ToolError) {
throw error;
}
throw new ToolError(error instanceof Error ? error.message : String(error));
} finally {
db?.close();
}
}
@@ -0,0 +1,199 @@
import * as path from "node:path";
import { type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives";
import { LRUCache } from "@oh-my-pi/pi-utils/lru";
import { isMarkdownPath } from "../modes/theme/theme";
import type { ToolSession } from "../sdk";
import { resolveFileDisplayMode } from "../utils/file-display-mode";
import {
canMergeBracePair,
countTextLines,
type ElidedRange,
formatMergedBraceLine,
formatSingleLine,
} from "./read-format";
import { throwIfAborted } from "./tool-errors";
// Per-session memo for tree-sitter summaries. `summarizeCode` is a pure function
// of (code, path, fold settings) but costs ~12-18ms for a ~1500-line file, and a
// repeat summary read of the same unchanged file re-parses from scratch. Key on
// the content hash of the freshly-read bytes (+ path + fold settings): the file
// is still read fresh on every call, so a hit only reuses the deterministic
// parse — there is no staleness window and no stat guard is needed. Bounded LRU,
// aged out with the session via WeakMap.
// Unusable results (not parsed, or nothing elided) are memoized as `false`: the
// full SummaryResult embeds the whole source in kept segments, and the caller
// only ever renders `parsed && elided` summaries — caching the segments would
// retain up to 48 near-2MiB sources just to remember "no summary".
const SUMMARY_CACHE_MAX = 48;
const summaryParseCaches = new WeakMap<object, LRUCache<string, SummaryResult | false>>();
function getSummaryParseCache(session: object): LRUCache<string, SummaryResult | false> {
let cache = summaryParseCaches.get(session);
if (!cache) {
cache = new LRUCache<string, SummaryResult | false>({ max: SUMMARY_CACHE_MAX });
summaryParseCaches.set(session, cache);
}
return cache;
}
const MAX_SUMMARY_BYTES = 2 * 1024 * 1024;
const MAX_SUMMARY_LINES = 20_000;
/**
* Prose files (Markdown flavors and plain text) skip code-block summarization
* unless `read.summarize.prose` opts them in.
*/
export function isProseSummaryPath(filePath: string): boolean {
return isMarkdownPath(filePath) || path.extname(filePath).toLowerCase() === ".txt";
}
export function routeReadThroughBridge(
session: ToolSession,
absolutePath: string,
options?: { line?: number; limit?: number },
): Promise<string> | undefined {
const bridge = session.getClientBridge?.();
if (!bridge?.capabilities.readTextFile || !bridge.readTextFile) return undefined;
return bridge.readTextFile({ path: absolutePath, ...options });
}
export async function trySummarize(
session: ToolSession,
absolutePath: string,
fileSize: number,
signal?: AbortSignal,
): Promise<SummaryResult | null> {
if (fileSize > MAX_SUMMARY_BYTES) return null;
try {
throwIfAborted(signal);
const bridgePromise = routeReadThroughBridge(session, absolutePath);
const code =
bridgePromise !== undefined
? await bridgePromise.catch(() => Bun.file(absolutePath).text())
: await Bun.file(absolutePath).text();
throwIfAborted(signal);
const lineCount = countTextLines(code);
if (lineCount > MAX_SUMMARY_LINES) return null;
if (lineCount < session.settings.get("read.summarize.minTotalLines")) return null;
const minBodyLines = session.settings.get("read.summarize.minBodyLines");
const minCommentLines = session.settings.get("read.summarize.minCommentLines");
const unfoldUntilLines = session.settings.get("read.summarize.unfoldUntil");
const unfoldLimitLines = session.settings.get("read.summarize.unfoldLimit");
const cache = getSummaryParseCache(session);
const cacheKey = `${absolutePath}\0${Bun.hash(code)}\0${minBodyLines},${minCommentLines},${unfoldUntilLines},${unfoldLimitLines}`;
const memoized = cache.get(cacheKey);
if (memoized !== undefined) return memoized || null;
const result = summarizeCode({
code,
path: absolutePath,
minBodyLines,
minCommentLines,
unfoldUntilLines,
unfoldLimitLines,
});
const usable = result.parsed && result.elided ? result : false;
cache.set(cacheKey, usable);
return usable || null;
} catch {
return null;
}
}
export function renderSummary(
session: ToolSession,
summary: SummaryResult,
): {
text: string;
displayText: string;
elidedRanges: ElidedRange[];
elidedLines: number;
} {
const displayMode = resolveFileDisplayMode(session);
const shouldAddHashLines = displayMode.hashLines;
const shouldAddLineNumbers = shouldAddHashLines ? false : displayMode.lineNumbers;
// Flatten segments into per-line units so we can merge a kept-head /
// elided / kept-tail sandwich into a single brace-pair line when the
// boundary lines look like `… {` and `}` (or matching variants).
type Unit =
| { kind: "line"; line: number; text: string }
| { kind: "elided"; startLine: number; endLine: number }
| {
kind: "merged";
startLine: number;
endLine: number;
headText: string;
tailText: string;
};
const raw: Unit[] = [];
for (const segment of summary.segments) {
if (segment.kind === "elided") {
raw.push({ kind: "elided", startLine: segment.startLine, endLine: segment.endLine });
continue;
}
const text = segment.text ?? "";
if (text.length === 0) continue;
const lines = text.split("\n");
for (let i = 0; i < lines.length; i++) {
raw.push({ kind: "line", line: segment.startLine + i, text: lines[i] });
}
}
const units: Unit[] = [];
let i = 0;
while (i < raw.length) {
const cur = raw[i];
if (cur.kind === "elided") {
const prev = units.length > 0 ? units[units.length - 1] : null;
const next = i + 1 < raw.length ? raw[i + 1] : null;
if (prev?.kind === "line" && next?.kind === "line" && canMergeBracePair(prev.text, next.text)) {
units.pop();
units.push({
kind: "merged",
startLine: prev.line,
endLine: next.line,
headText: prev.text,
tailText: next.text,
});
i += 2;
continue;
}
}
units.push(cur);
i++;
}
const modelParts: string[] = [];
const displayParts: string[] = [];
const elidedRanges: ElidedRange[] = [];
let elidedLines = 0;
for (const unit of units) {
if (unit.kind === "elided") {
modelParts.push("…");
displayParts.push("…");
elidedRanges.push({ start: unit.startLine, end: unit.endLine });
elidedLines += unit.endLine - unit.startLine + 1;
continue;
}
if (unit.kind === "merged") {
const formatted = formatMergedBraceLine(
unit.startLine,
unit.endLine,
unit.headText,
unit.tailText,
shouldAddHashLines,
shouldAddLineNumbers,
);
modelParts.push(formatted.model);
displayParts.push(formatted.display);
// Suggest the full brace range so re-reading shows both braces
// plus the elided body in one shot.
elidedRanges.push({ start: unit.startLine, end: unit.endLine });
// Merged brace pair encloses (start+1)..(end-1) as elided.
elidedLines += Math.max(0, unit.endLine - unit.startLine - 1);
continue;
}
modelParts.push(formatSingleLine(unit.line, unit.text, shouldAddHashLines, shouldAddLineNumbers));
displayParts.push(unit.text);
}
return { text: modelParts.join("\n"), displayText: displayParts.join("\n"), elidedRanges, elidedLines };
}
File diff suppressed because it is too large Load Diff