Files
oh-my-pi/packages/hashline/src/input.ts
T
can1357 01c34db450 feat(hashline): added full-file hash snapshots with 4-hex tags
- Replaced snapshot internals with full-file records and removed contiguous/sparse snapshot APIs.
- Added file-hash normalization, computed `computeFileHash`, and updated grammar/messages to 4-hex tags.
- Simplified recovery by checking whole-file hashes first, then applying merge-replay fallback after mismatches.
- Updated coding-agent tools to use `record`/`recordFileSnapshot` and skip hash headers for unsnapshotted large files.
- Expanded patcher and snapshot tests to verify 4-hex anchors, hash deduplication, and cache-capped behavior.
2026-05-29 18:12:08 +02:00

388 lines
15 KiB
TypeScript

/**
* Top-level patch parser. Splits an authored hashline input into a list of
* {@link PatchSection}s, each rooted at a `¶PATH#HASH` header, then exposes
* a {@link Patch} class that gives lazy access to the parsed edits per
* section.
*
* The splitter is purely lexical — it doesn't know whether a section's path
* actually exists. That's the patcher's job.
*/
import * as path from "node:path";
import { applyEdits } from "./apply";
import { HL_FILE_HASH_LENGTH, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format";
import { parsePatch, parsePatchStreaming } from "./parser";
import { Tokenizer } from "./tokenizer";
import type { ApplyResult, Edit, SplitOptions } from "./types";
// Pure classification — single shared tokenizer is safe.
const TOKENIZER = new Tokenizer();
function unquoteHashlinePath(pathText: string): string {
if (pathText.length < 2) return pathText;
const first = pathText[0];
const last = pathText[pathText.length - 1];
if ((first === '"' || first === "'") && first === last) return pathText.slice(1, -1);
return pathText;
}
/**
* Strip apply_patch-style noise that models reflexively prepend to the
* path. Examples observed in benchmark traces:
*
* `Update File:foo.ts`, `Update:foo.ts`, `UpdateFile:foo.ts`,
* `Update/File:foo.ts`, `Update-file:foo.ts`, `Update(File):foo.ts`,
* `Update<File:foo.ts`, `Add File:foo.ts`, `Delete File:foo.ts`,
* `Move to:foo.ts`, `***foo.ts`, `***Update File:foo.ts`.
*
* We strip a leading `***` (the model duplicating the header sigil) and a
* leading `(Update|Add|Delete|Move)[<separator>]*(File|to)?[<separator>]*:`
* keyword block, case-insensitive. The remaining text is the real path.
*/
const APPLY_PATCH_PATH_NOISE_RE =
/^\*{0,3}\s*(?:(?:update|add|delete|move)[^A-Za-z0-9]*(?:file|to)?[^A-Za-z0-9]*:)?\s*\*{0,3}\s*/i;
function stripApplyPatchPathNoise(pathText: string): string {
return pathText.replace(APPLY_PATCH_PATH_NOISE_RE, "");
}
/**
* Best-effort recovery for `¶`-prefixed lines the strict tokenizer
* rejects. Strips apply_patch keyword noise (`Update File:`, `Update:`,
* etc.) and an extra leading `***` (some models emit a hybrid `¶***foo.ts`
* shape), then expects `PATH(#HASH)?` with no embedded whitespace.
* Returns `null` when no clean path can be salvaged.
*/
function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null {
if (!line.startsWith(HL_FILE_PREFIX)) return null;
const body = stripApplyPatchPathNoise(line.slice(HL_FILE_PREFIX.length).trim());
if (body.length === 0) return null;
const match = new RegExp(`^(\\S+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body);
if (match === null) return null;
const path = normalizeHashlinePath(match[1], cwd);
if (path.length === 0) return null;
return match[2] !== undefined ? { path, fileHash: match[2].toUpperCase(), diff: "" } : { path, diff: "" };
}
function normalizeHashlinePath(rawPath: string, cwd?: string): string {
const unquoted = stripApplyPatchPathNoise(unquoteHashlinePath(rawPath.trim()));
if (!cwd || !path.isAbsolute(unquoted)) return unquoted;
const relative = path.relative(path.resolve(cwd), path.resolve(unquoted));
const isWithinCwd = relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative));
return isWithinCwd ? relative || "." : unquoted;
}
interface RawSection {
path: string;
fileHash?: string;
diff: string;
}
/**
* Parse a `¶PATH[#hash]` header line. Returns `null` for lines that do
* not start with `¶`. Throws the strict "Input header must be …" error
* when a `¶`-prefixed line fails the strict shape (so malformed paths
* surface immediately instead of being silently re-classified as payload).
*/
function parseHashlineHeaderLine(line: string, cwd?: string): RawSection | null {
const trimmed = line.trimEnd();
if (!trimmed.startsWith(HL_FILE_PREFIX)) return null;
const token = TOKENIZER.tokenize(trimmed);
if (token.kind !== "header") {
// Recovery: try to extract a path from the raw line after stripping
// apply_patch noise. This handles `*** Update File:foo.ts#CB5` and
// the half-dozen variants models actually emit.
const recovered = tryParseRecoveryHeader(trimmed, cwd);
if (recovered !== null) return recovered;
throw new Error(
`Input header must be ${HL_FILE_PREFIX}PATH or ${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}TAG with a ${HL_FILE_HASH_LENGTH}-hex content-hash tag; got ${JSON.stringify(trimmed)}.`,
);
}
const parsedPath = normalizeHashlinePath(token.path, cwd);
if (parsedPath.length === 0) {
throw new Error(`Input header "${HL_FILE_PREFIX}" is empty; provide a file path.`);
}
return token.fileHash !== undefined
? { path: parsedPath, fileHash: token.fileHash, diff: "" }
: { path: parsedPath, diff: "" };
}
function stripLeadingBlankLines(input: string): string {
const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input;
const lines = stripped.split("\n");
while (lines.length > 0) {
const head = lines[0].replace(/\r$/, "");
if (head.trim().length === 0 || TOKENIZER.tokenize(head).kind === "envelope-begin") {
lines.shift();
continue;
}
break;
}
return lines.join("\n");
}
/**
* Returns true when the input contains at least one line that the tokenizer
* recognizes as a hashline op. Used by streaming previews to decide whether
* the partial input is worth treating as a hashline patch yet.
*/
export function containsRecognizableHashlineOperations(input: string): boolean {
for (const line of input.split(/\r?\n/)) {
if (TOKENIZER.isOp(line)) return true;
}
return false;
}
function normalizeFallbackInput(input: string, options: SplitOptions): string {
const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input;
const hasExplicitHeader = stripped
.split(/\r?\n/)
.some(rawLine => parseHashlineHeaderLine(rawLine, options.cwd) !== null);
if (hasExplicitHeader) return input;
if (!options.path || !containsRecognizableHashlineOperations(input)) return input;
const fallbackPath = normalizeHashlinePath(options.path, options.cwd);
if (fallbackPath.length === 0) return input;
return `${HL_FILE_PREFIX}${fallbackPath}\n${input}`;
}
function splitRawSections(input: string, options: SplitOptions = {}): RawSection[] {
const stripped = stripLeadingBlankLines(normalizeFallbackInput(input, options));
const lines = stripped.split(/\r?\n/);
const firstLine = lines[0] ?? "";
if (parseHashlineHeaderLine(firstLine, options.cwd) === null) {
// Catch unified-diff hunk-header contamination on the first line so
// the model sees a focused error.
const firstTrimmed = firstLine.trimEnd();
if (/^@@\s+[-+]?\d+,\d+\s+[-+]?\d+,\d+\s+@@/.test(firstTrimmed)) {
throw new Error(
"unified-diff hunk header (`@@ -N,M +N,M @@`) is not valid in hashline. " +
"File sections start with `¶path#HASH`; use `replace`, `delete`, or `insert` ops.",
);
}
const preview = JSON.stringify(firstLine.slice(0, 120));
throw new Error(
`input must begin with "${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}HASH" on the first non-blank line for anchored edits; got: ${preview}. ` +
`Example: "${HL_FILE_PREFIX}src/foo.ts${HL_FILE_HASH_SEP}0A3" then edit ops.`,
);
}
const sections: RawSection[] = [];
let current: RawSection | undefined;
let currentLines: string[] = [];
const flush = () => {
if (!current) return;
const hasOps = currentLines.some(line => line.trim().length > 0);
if (hasOps) sections.push({ ...current, diff: currentLines.join("\n") });
currentLines = [];
};
for (const line of lines) {
const trimmed = line.trimEnd();
const token = TOKENIZER.tokenize(line);
if (token.kind === "envelope-end" || token.kind === "abort") break;
if (token.kind === "envelope-begin") continue;
// Route every `¶`-prefixed line through parseHashlineHeaderLine so
// malformed headers still raise the strict "Input header must be …"
// diagnostic (the tokenizer alone would silently classify them as
// payload).
if (trimmed.startsWith(HL_FILE_PREFIX)) {
const header = parseHashlineHeaderLine(line, options.cwd);
if (header !== null) {
flush();
current = header;
currentLines = [];
continue;
}
}
currentLines.push(line);
}
flush();
return sections;
}
/**
* Snapshot of one section in a parsed {@link Patch}: a target file plus the
* lazily-parsed list of edits that should land on it. Constructed by
* {@link Patch.parse}; consumers usually iterate `patch.sections` rather
* than build these directly.
*/
export class PatchSection {
readonly path: string;
readonly fileHash: string | undefined;
readonly diff: string;
#parsed: { edits: Edit[]; warnings: string[] } | undefined;
constructor(raw: RawSection) {
this.path = raw.path;
this.fileHash = raw.fileHash;
this.diff = raw.diff;
}
/**
* Parse this section's diff body. Cached: subsequent calls return the
* same `{ edits, warnings }` object so callers can safely call this from
* multiple paths (preflight, apply, diff-preview).
*/
parse(): { edits: Edit[]; warnings: readonly string[] } {
this.#parsed ??= parsePatch(this.diff);
return this.#parsed;
}
/** Parsed edits for this section. */
get edits(): readonly Edit[] {
return this.parse().edits;
}
/** Warnings emitted during parsing of this section. */
get warnings(): readonly string[] {
return this.parse().warnings;
}
/**
* True when at least one edit anchors to concrete file content. Pure
* `insert head:` / `insert tail:` literal inserts do not count: those are
* safe to apply to files that don't yet exist.
*/
get hasAnchorScopedEdit(): boolean {
return this.edits.some(edit => {
if (edit.kind === "delete") return true;
return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor";
});
}
/** Anchor lines touched by this section, sorted ascending and deduplicated. */
collectAnchorLines(): readonly number[] {
const lines = new Set<number>();
for (const edit of this.edits) {
if (edit.kind === "delete") {
lines.add(edit.anchor.line);
continue;
}
if (edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor") {
lines.add(edit.cursor.anchor.line);
}
}
return [...lines].sort((a, b) => a - b);
}
/**
* Apply this section's edits to `text` and return the post-edit result.
* Pure: does no I/O, does not validate the section snapshot tag. The
* {@link Patcher} owns tag validation and recovery; reach for this
* method directly when you've already validated the file content and
* just want the result.
*/
applyTo(text: string): ApplyResult {
const { edits, warnings } = this.parse();
const result = applyEdits(text, [...edits]);
// Preserve parse warnings so consumers don't need to call `parse()`
// separately.
const merged = warnings.length === 0 ? result.warnings : [...warnings, ...(result.warnings ?? [])];
return merged && merged.length > 0
? { ...result, warnings: merged }
: { text: result.text, firstChangedLine: result.firstChangedLine };
}
/**
* Streaming-tolerant counterpart to {@link applyTo}. Uses
* {@link parsePatchStreaming} so a trailing in-flight op (no payload yet,
* or a per-token parse error mid-stream) does not throw or emit a phantom
* empty-payload edit. Intended for incremental diff previews; the writer
* path should always use {@link applyTo}.
*/
applyPartialTo(text: string): ApplyResult {
const { edits, warnings } = parsePatchStreaming(this.diff);
const result = applyEdits(text, [...edits]);
const merged = warnings.length === 0 ? result.warnings : [...warnings, ...(result.warnings ?? [])];
return merged && merged.length > 0
? { ...result, warnings: merged }
: { text: result.text, firstChangedLine: result.firstChangedLine };
}
}
/**
* A parsed hashline patch — zero or more {@link PatchSection}s, each rooted
* at a `¶PATH#HASH` header. Construct via {@link Patch.parse}.
*
* `Patch` is pure data: parsing is line-anchored and does not look at the
* filesystem. To apply a patch, hand it to {@link Patcher.apply}.
*/
export class Patch {
readonly sections: readonly PatchSection[];
private constructor(sections: PatchSection[]) {
this.sections = sections;
}
/**
* Parse `input` into a {@link Patch}. `options.cwd` resolves absolute
* paths inside headers to cwd-relative form; `options.path` provides a
* fallback when the input lacks a header but contains hashline ops
* (useful for streaming previews).
*
* Consecutive sections targeting the same path are merged into a single
* section with concatenated diff bodies. Anchors authored against the
* same file snapshot must be applied as one batch; otherwise the first
* sub-edit shifts line numbers out from under the second's anchors and
* validation fails.
*/
static parse(input: string, options: SplitOptions = {}): Patch {
const raw = mergeSamePathSections(splitRawSections(input, options));
return new Patch(raw.map(section => new PatchSection(section)));
}
/**
* Parse `input` and return only the first section. Throws if the input
* has zero sections. Convenience for the single-section case where the
* caller already knows the patch is one hunk.
*/
static parseSingle(input: string, options: SplitOptions = {}): PatchSection {
const patch = Patch.parse(input, options);
const first = patch.sections[0];
if (!first) throw new Error("Patch input did not produce any sections.");
return first;
}
}
/**
* Collapse consecutive or interleaved sections targeting the same path into a
* single section with concatenated diffs. Anchors authored against the same
* file snapshot must be applied as one batch; otherwise the first sub-edit
* shifts line numbers out from under the second's anchors and validation
* fails. Path order is preserved by first occurrence.
*/
function mergeSamePathSections(sections: RawSection[]): RawSection[] {
const byPath = new Map<string, { fileHash?: string; diffs: string[] }>();
for (const section of sections) {
const existing = byPath.get(section.path);
if (existing) {
if (
existing.fileHash !== undefined &&
section.fileHash !== undefined &&
existing.fileHash !== section.fileHash
) {
throw new Error(
`Conflicting hashline snapshot tags for ${section.path}: #${existing.fileHash} and #${section.fileHash}. Re-read the file and retry with one current header.`,
);
}
if (existing.fileHash === undefined && section.fileHash !== undefined) existing.fileHash = section.fileHash;
existing.diffs.push(section.diff);
continue;
}
byPath.set(section.path, {
...(section.fileHash !== undefined ? { fileHash: section.fileHash } : {}),
diffs: [section.diff],
});
}
return Array.from(byPath, ([sectionPath, entry]) => ({
path: sectionPath,
...(entry.fileHash !== undefined ? { fileHash: entry.fileHash } : {}),
diff: entry.diffs.join("\n"),
}));
}