From b12e4698a69b6cf2c48feb994fdb7706b9603524 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 27 May 2026 04:02:04 +0200 Subject: [PATCH] feat: added @oh-my-pi/hashline package and migrated hashline tooling - Added a dedicated @oh-my-pi/hashline package with parser, patcher, filesystem, snapshots, and release metadata. - Migrated coding-agent hashline and stream entrypoints to @oh-my-pi/hashline and removed old hashline module exports. - Changed multi-section hashline execution to validate section hashes and flush diagnostics only at the final commit. - Added session fileSnapshotStore support and rewired edit/read/search/write tools to use it instead of fileReadCache. --- bun.lock | 15 + docs/tools/edit.md | 2 +- package.json | 1 + packages/coding-agent/CHANGELOG.md | 6 + packages/coding-agent/package.json | 1 + .../coding-agent/src/edit/file-read-cache.ts | 138 ---- .../src/edit/file-snapshot-store.ts | 22 + .../coding-agent/src/edit/hashline/diff.ts | 88 +++ .../coding-agent/src/edit/hashline/execute.ts | 174 +++++ .../src/edit/hashline/filesystem.ts | 125 ++++ .../coding-agent/src/edit/hashline/index.ts | 4 + .../coding-agent/src/edit/hashline/params.ts | 11 + packages/coding-agent/src/edit/index.ts | 22 +- packages/coding-agent/src/edit/normalize.ts | 52 +- packages/coding-agent/src/edit/renderer.ts | 2 +- packages/coding-agent/src/edit/streaming.ts | 17 +- .../coding-agent/src/hashline/bigrams.json | 649 ------------------ packages/coding-agent/src/hashline/diff.ts | 82 --- packages/coding-agent/src/hashline/execute.ts | 334 --------- packages/coding-agent/src/hashline/input.ts | 137 ---- .../coding-agent/src/hashline/recovery.ts | 139 ---- packages/coding-agent/src/hashline/types.ts | 66 -- packages/coding-agent/src/index.ts | 1 - packages/coding-agent/src/tools/ast-edit.ts | 2 +- packages/coding-agent/src/tools/ast-grep.ts | 6 +- packages/coding-agent/src/tools/index.ts | 11 +- packages/coding-agent/src/tools/read.ts | 10 +- packages/coding-agent/src/tools/search.ts | 20 +- packages/coding-agent/src/tools/write.ts | 6 +- .../coding-agent/src/utils/file-mentions.ts | 2 +- .../coding-agent/test/core/hashline.test.ts | 98 ++- packages/hashline/CHANGELOG.md | 15 + packages/hashline/README.md | 78 +++ packages/hashline/package.json | 61 ++ .../src/hashline => hashline/src}/apply.ts | 135 ++-- .../hashline => hashline/src}/diff-preview.ts | 25 +- .../hash.ts => hashline/src/format.ts} | 91 +-- packages/hashline/src/fs.ts | 159 +++++ .../hashline => hashline/src}/grammar.lark | 10 +- .../src/hashline => hashline/src}/index.ts | 14 +- packages/hashline/src/input.ts | 319 +++++++++ .../constants.ts => hashline/src/messages.ts} | 46 +- .../anchors.ts => hashline/src/mismatch.ts} | 43 +- packages/hashline/src/normalize.ts | 39 ++ .../executor.ts => hashline/src/parser.ts} | 99 +-- packages/hashline/src/patcher.ts | 343 +++++++++ .../src/hashline => hashline/src}/prefixes.ts | 41 +- .../hashline.md => hashline/src/prompt.md} | 0 packages/hashline/src/recovery.ts | 163 +++++ packages/hashline/src/snapshots.ts | 171 +++++ .../src/hashline => hashline/src}/stream.ts | 29 +- .../hashline => hashline/src}/tokenizer.ts | 87 ++- packages/hashline/src/types.ts | 87 +++ packages/hashline/tsconfig.json | 7 + packages/hashline/tsconfig.publish.json | 22 + packages/natives/CHANGELOG.md | 6 + packages/natives/native/index.d.ts | 350 ++++++++++ packages/natives/native/index.js | 24 + scripts/ci-release-publish.ts | 1 + 59 files changed, 2768 insertions(+), 1940 deletions(-) delete mode 100644 packages/coding-agent/src/edit/file-read-cache.ts create mode 100644 packages/coding-agent/src/edit/file-snapshot-store.ts create mode 100644 packages/coding-agent/src/edit/hashline/diff.ts create mode 100644 packages/coding-agent/src/edit/hashline/execute.ts create mode 100644 packages/coding-agent/src/edit/hashline/filesystem.ts create mode 100644 packages/coding-agent/src/edit/hashline/index.ts create mode 100644 packages/coding-agent/src/edit/hashline/params.ts delete mode 100644 packages/coding-agent/src/hashline/bigrams.json delete mode 100644 packages/coding-agent/src/hashline/diff.ts delete mode 100644 packages/coding-agent/src/hashline/execute.ts delete mode 100644 packages/coding-agent/src/hashline/input.ts delete mode 100644 packages/coding-agent/src/hashline/recovery.ts delete mode 100644 packages/coding-agent/src/hashline/types.ts create mode 100644 packages/hashline/CHANGELOG.md create mode 100644 packages/hashline/README.md create mode 100644 packages/hashline/package.json rename packages/{coding-agent/src/hashline => hashline/src}/apply.ts (87%) rename packages/{coding-agent/src/hashline => hashline/src}/diff-preview.ts (51%) rename packages/{coding-agent/src/hashline/hash.ts => hashline/src/format.ts} (67%) create mode 100644 packages/hashline/src/fs.ts rename packages/{coding-agent/src/hashline => hashline/src}/grammar.lark (57%) rename packages/{coding-agent/src/hashline => hashline/src}/index.ts (50%) create mode 100644 packages/hashline/src/input.ts rename packages/{coding-agent/src/hashline/constants.ts => hashline/src/messages.ts} (59%) rename packages/{coding-agent/src/hashline/anchors.ts => hashline/src/mismatch.ts} (69%) create mode 100644 packages/hashline/src/normalize.ts rename packages/{coding-agent/src/hashline/executor.ts => hashline/src/parser.ts} (83%) create mode 100644 packages/hashline/src/patcher.ts rename packages/{coding-agent/src/hashline => hashline/src}/prefixes.ts (72%) rename packages/{coding-agent/src/prompts/tools/hashline.md => hashline/src/prompt.md} (100%) create mode 100644 packages/hashline/src/recovery.ts create mode 100644 packages/hashline/src/snapshots.ts rename packages/{coding-agent/src/hashline => hashline/src}/stream.ts (77%) rename packages/{coding-agent/src/hashline => hashline/src}/tokenizer.ts (86%) create mode 100644 packages/hashline/src/types.ts create mode 100644 packages/hashline/tsconfig.json create mode 100644 packages/hashline/tsconfig.publish.json diff --git a/bun.lock b/bun.lock index 6c617aaed..a646f4cce 100644 --- a/bun.lock +++ b/bun.lock @@ -53,6 +53,7 @@ "@agentclientprotocol/sdk": "catalog:", "@babel/parser": "catalog:", "@mozilla/readability": "catalog:", + "@oh-my-pi/hashline": "catalog:", "@oh-my-pi/omp-stats": "catalog:", "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", @@ -78,6 +79,17 @@ "@types/bun": "catalog:", }, }, + "packages/hashline": { + "name": "@oh-my-pi/hashline", + "version": "15.5.3", + "dependencies": { + "diff": "catalog:", + "lru-cache": "catalog:", + }, + "devDependencies": { + "@types/bun": "catalog:", + }, + }, "packages/natives": { "name": "@oh-my-pi/pi-natives", "version": "15.5.3", @@ -209,6 +221,7 @@ "@bufbuild/protoc-gen-es": "^2.12.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.6.2", + "@oh-my-pi/hashline": "15.5.3", "@oh-my-pi/omp-stats": "15.5.3", "@oh-my-pi/pi-agent-core": "15.5.3", "@oh-my-pi/pi-ai": "15.5.3", @@ -562,6 +575,8 @@ "@octokit/types": ["@octokit/types@16.0.0", "", { "dependencies": { "@octokit/openapi-types": "^27.0.0" } }, "sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg=="], + "@oh-my-pi/hashline": ["@oh-my-pi/hashline@workspace:packages/hashline"], + "@oh-my-pi/omp-stats": ["@oh-my-pi/omp-stats@workspace:packages/stats"], "@oh-my-pi/pi-agent-core": ["@oh-my-pi/pi-agent-core@workspace:packages/agent"], diff --git a/docs/tools/edit.md b/docs/tools/edit.md index f9024e6d1..d433bd98e 100644 --- a/docs/tools/edit.md +++ b/docs/tools/edit.md @@ -71,7 +71,7 @@ Warnings: - While the model is still typing arguments, the TUI can compute a diff preview with `packages/coding-agent/src/edit/streaming.ts`; that preview is not a deferred action and does not block execution. ## Flow -1. `EditTool.execute()` in `packages/coding-agent/src/edit/index.ts` resolves the active mode. Default is `hashline`; `customFormat` exposes `packages/coding-agent/src/hashline/grammar.lark` with `$HFILE_HASH$` / `$HOP_INSERT_BEFORE$` / `$HOP_INSERT_AFTER$` / `$HOP_REPLACE$` / `$HOP_CHARS$` / `$HFILE$` placeholders filled from `packages/coding-agent/src/hashline/hash.ts`. +1. `EditTool.execute()` in `packages/coding-agent/src/edit/index.ts` resolves the active mode. Default is `hashline`; `customFormat` exposes `packages/hashline/src/grammar.lark` as a constant string with op sigils and the section-header `¶` inlined. 2. `executeHashlineSingle()` in `packages/coding-agent/src/hashline/execute.ts` splits the raw `input` into `¶PATH#HASH` / `¶PATH` sections with `splitHashlineInputs()`. 3. If multiple sections target the same path, `mergeSamePathSections()` concatenates them before execution so every op still refers to the original file snapshot. 4. Multi-section calls run a preflight pass (`preflightHashlineSection()`): parse ops, enforce plan-mode write rules, load the current file, reject anchor-scoped edits against missing files, reject auto-generated files, apply edits in memory, and fail if the result is a no-op. This prevents partial batches. diff --git a/package.json b/package.json index 5c1a710ad..cbcdc34d1 100644 --- a/package.json +++ b/package.json @@ -20,6 +20,7 @@ "@bufbuild/protoc-gen-es": "^2.12.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.6.2", + "@oh-my-pi/hashline": "15.5.3", "@oh-my-pi/omp-stats": "15.5.3", "@oh-my-pi/pi-agent-core": "15.5.3", "@oh-my-pi/pi-ai": "15.5.3", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e248a2bef..1a2497a1c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,11 @@ # Changelog ## [Unreleased] + +### Breaking Changes + +- Removed the package root `hashline` export so imports from the top-level entrypoint can no longer access `hashline` helpers directly + ### Added - Added `read.summarize.minTotalLines` setting (default 100) to set the minimum file length that triggers read summarization @@ -8,6 +13,7 @@ ### Changed +- Changed multi-section hashline `edit` execution to defer LSP diagnostics flushing until the final section is written - Changed read to return verbatim contents for files shorter than `read.summarize.minTotalLines` instead of summarizing them - Changed `search` path line-range filtering to include only matches and context lines that fall inside the requested ranges diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 08a45ad9a..47b50752c 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -47,6 +47,7 @@ "@agentclientprotocol/sdk": "catalog:", "@babel/parser": "catalog:", "@mozilla/readability": "catalog:", + "@oh-my-pi/hashline": "catalog:", "@oh-my-pi/omp-stats": "catalog:", "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", diff --git a/packages/coding-agent/src/edit/file-read-cache.ts b/packages/coding-agent/src/edit/file-read-cache.ts deleted file mode 100644 index 54a2f1897..000000000 --- a/packages/coding-agent/src/edit/file-read-cache.ts +++ /dev/null @@ -1,138 +0,0 @@ -/** - * Per-session cache of file contents as they were rendered to the model by - * the `read` and `search` tools in the current agent session. - * - * Used by hashline-mode anchor-stale recovery: if the model authored anchors - * against a version of the file that no longer matches what is on disk — - * because a subagent, the user, a linter, or a formatter modified the file - * between the read and the edit — we replay the edits against the cached - * pre-edit snapshot and 3-way-merge the result onto the live file. - * - * Scoped per `ToolSession`: the cache lives on the session object itself, so - * different sessions never share snapshots and entries get reclaimed when - * the session goes out of scope. Each session keeps a small LRU window of - * paths; each path keeps a short ring of recent snapshots so follow-up edits - * can recover from the agent's own prior writes as well as stale reads. - */ -import { LRUCache } from "lru-cache/raw"; -import type { ToolSession } from "../tools"; - -const MAX_PATHS_PER_SESSION = 30; -const MAX_SNAPSHOTS_PER_PATH = 4; - -export interface FileReadSnapshot { - /** 1-indexed line number → exact line content as observed by `read`/`search`. */ - lines: Map; - /** Full normalized text when the read path observed the whole file. */ - fullText?: string; - /** 4-hex hash of `fullText`, or a sparse snapshot hash supplied by search. */ - fileHash?: string; - recordedAt: number; -} - -interface FileReadSnapshotMetadata { - fullText?: string; - fileHash?: string; -} - -export class FileReadCache { - #snapshots = new LRUCache({ max: MAX_PATHS_PER_SESSION }); - - /** Look up the most recent snapshot for `absPath`, or `null` if absent. */ - get(absPath: string): FileReadSnapshot | null { - return this.#snapshots.get(absPath)?.[0] ?? null; - } - - /** Look up the most recent snapshot for `absPath` whose file hash matches. */ - getByHash(absPath: string, fileHash: string): FileReadSnapshot | null { - const history = this.#snapshots.get(absPath); - return history?.find(snapshot => snapshot.fileHash === fileHash) ?? null; - } - - /** Record a contiguous run of lines (e.g. from a `read` tool). `startLine` is 1-indexed. */ - recordContiguous( - absPath: string, - startLine: number, - lines: readonly string[], - metadata: FileReadSnapshotMetadata = {}, - ): void { - if (lines.length === 0 && metadata.fullText === undefined) return; - const entries: Array = lines.map((line, idx) => [startLine + idx, line] as const); - this.#record(absPath, entries, metadata); - } - - /** Record sparse `(lineNumber, content)` pairs (e.g. `search` matches plus context). */ - recordSparse( - absPath: string, - entries: Iterable, - metadata: FileReadSnapshotMetadata = {}, - ): void { - const arr = Array.from(entries); - if (arr.length === 0 && metadata.fullText === undefined) return; - this.#record(absPath, arr, metadata); - } - - /** Drop the snapshot history for a single path. */ - invalidate(absPath: string): void { - this.#snapshots.delete(absPath); - } - - /** Drop every snapshot history. */ - clear(): void { - this.#snapshots.clear(); - } - - #record( - absPath: string, - entries: ReadonlyArray, - metadata: FileReadSnapshotMetadata, - ): void { - const history = this.#snapshots.get(absPath) ?? []; - const head = history[0]; - const now = Date.now(); - if (head && !hasConflict(head.lines, entries) && !hasHashConflict(head, metadata)) { - for (const [lineNum, content] of entries) head.lines.set(lineNum, content); - if (metadata.fullText !== undefined) head.fullText = metadata.fullText; - if (metadata.fileHash !== undefined) head.fileHash = metadata.fileHash; - head.recordedAt = now; - // `get` above already touched LRU recency for this key. - return; - } - - const nextSnapshot: FileReadSnapshot = { - lines: new Map(entries), - ...metadata, - recordedAt: now, - }; - const dedupedHistory = history.filter(snapshot => !isSameSnapshotIdentity(snapshot, nextSnapshot)); - this.#snapshots.set(absPath, [nextSnapshot, ...dedupedHistory].slice(0, MAX_SNAPSHOTS_PER_PATH)); - } -} - -function hasConflict(existing: Map, incoming: ReadonlyArray): boolean { - for (const [lineNum, content] of incoming) { - const prior = existing.get(lineNum); - if (prior !== undefined && prior !== content) return true; - } - return false; -} - -function hasHashConflict(existing: FileReadSnapshot, metadata: FileReadSnapshotMetadata): boolean { - return metadata.fileHash !== undefined && existing.fileHash !== undefined && metadata.fileHash !== existing.fileHash; -} - -function isSameSnapshotIdentity(left: FileReadSnapshot, right: FileReadSnapshot): boolean { - if (left.fileHash !== undefined && right.fileHash !== undefined) return left.fileHash === right.fileHash; - if (left.fullText !== undefined && right.fullText !== undefined) return left.fullText === right.fullText; - return false; -} - -/** - * Look up (or lazily create) the file-read cache attached to a session. The - * cache is stored as `session.fileReadCache` so it lives exactly as long as - * the session itself. - */ -export function getFileReadCache(session: ToolSession): FileReadCache { - if (!session.fileReadCache) session.fileReadCache = new FileReadCache(); - return session.fileReadCache; -} diff --git a/packages/coding-agent/src/edit/file-snapshot-store.ts b/packages/coding-agent/src/edit/file-snapshot-store.ts new file mode 100644 index 000000000..55c839a10 --- /dev/null +++ b/packages/coding-agent/src/edit/file-snapshot-store.ts @@ -0,0 +1,22 @@ +/** + * Session-bound file snapshot store. + * + * Used by `read` and `search` to record exactly what the model saw, and by + * the hashline patcher to recover from stale section hashes (file changed + * externally between read and edit, or a prior in-session edit advanced + * the hash). The store is the {@link InMemorySnapshotStore} implementation + * from `@oh-my-pi/hashline`; the only coding-agent-specific concern here + * is wiring it onto the per-session {@link ToolSession} object. + */ +import { InMemorySnapshotStore } from "@oh-my-pi/hashline"; +import type { ToolSession } from "../tools"; + +/** + * Look up (or lazily create) the file snapshot store attached to a session. + * Storage lives on `session.fileSnapshotStore` so it ages out exactly with + * the session itself. + */ +export function getFileSnapshotStore(session: ToolSession): InMemorySnapshotStore { + if (!session.fileSnapshotStore) session.fileSnapshotStore = new InMemorySnapshotStore(); + return session.fileSnapshotStore; +} diff --git a/packages/coding-agent/src/edit/hashline/diff.ts b/packages/coding-agent/src/edit/hashline/diff.ts new file mode 100644 index 000000000..3979c10c5 --- /dev/null +++ b/packages/coding-agent/src/edit/hashline/diff.ts @@ -0,0 +1,88 @@ +/** + * Read-only hashline diff preview helpers used by the streaming edit + * renderer. Reads the target file, parses + applies the section's edits in + * memory (no FS write, no LSP writethrough), then hands the before/after + * pair to {@link generateDiffString} so the renderer can show the diff + * while the tool call is still streaming. + * + * Validation is intentionally light: only the section file hash is checked + * (so the preview goes red when anchors are stale), no plan-mode guards + * and no auto-generated-file refusal — those belong on the write path. + */ +import { + applyEdits, + computeFileHash, + Patch as HashlinePatch, + normalizeToLF, + type Patch, + type PatchSection, + stripBom, +} from "@oh-my-pi/hashline"; +import { resolveToCwd } from "../../tools/path-utils"; +import { generateDiffString } from "../diff"; +import { readEditFileText } from "../read-file"; + +export interface HashlineDiffOptions { + autoDropPureInsertDuplicates?: boolean; +} + +async function readSectionText(absolutePath: string, sectionPath: string): Promise { + try { + return await readEditFileText(absolutePath, sectionPath); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + throw new Error(message || `Unable to read ${sectionPath}`); + } +} + +function hasAnchorScoped(section: PatchSection): boolean { + return section.hasAnchorScopedEdit; +} + +function validateSectionHash(section: PatchSection, text: string): string | null { + if (section.fileHash === undefined) { + return hasAnchorScoped(section) + ? `Missing hashline file hash for anchored edit to ${section.path}; use \`¶${section.path}#hash\` from your latest read.` + : null; + } + const currentHash = computeFileHash(text); + if (currentHash === section.fileHash) return null; + return `Hashline file hash mismatch for ${section.path}: section is bound to #${section.fileHash}, but current file hashes to #${currentHash}; re-read and try again.`; +} + +export async function computeHashlineSectionDiff( + section: PatchSection, + cwd: string, + options: HashlineDiffOptions = {}, +): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> { + try { + const absolutePath = resolveToCwd(section.path, cwd); + const rawContent = await readSectionText(absolutePath, section.path); + const { text: content } = stripBom(rawContent); + const normalized = normalizeToLF(content); + const hashError = validateSectionHash(section, normalized); + if (hashError) return { error: hashError }; + const result = applyEdits(normalized, [...section.edits], options); + if (normalized === result.text) return { error: `No changes would be made to ${section.path}.` }; + return generateDiffString(normalized, result.text); + } catch (err) { + return { error: err instanceof Error ? err.message : String(err) }; + } +} + +export async function computeHashlineDiff( + input: { input: string; path?: string }, + cwd: string, + options: HashlineDiffOptions = {}, +): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> { + let patch: Patch; + try { + patch = HashlinePatch.parse(input.input, { cwd, path: input.path }); + } catch (err) { + return { error: err instanceof Error ? err.message : String(err) }; + } + if (patch.sections.length !== 1) { + return { error: "Streaming diff preview supports exactly one hashline section." }; + } + return computeHashlineSectionDiff(patch.sections[0], cwd, options); +} diff --git a/packages/coding-agent/src/edit/hashline/execute.ts b/packages/coding-agent/src/edit/hashline/execute.ts new file mode 100644 index 000000000..1f37fb86b --- /dev/null +++ b/packages/coding-agent/src/edit/hashline/execute.ts @@ -0,0 +1,174 @@ +/** + * Coding-agent runner that drives the hashline {@link Patcher} on behalf of + * the `edit` tool. Converts a `{input, path?}` tool-call payload into a + * fully-applied patch, wraps the result in the agent's + * {@link AgentToolResult} shape, and attaches LSP diagnostics + `outputMeta` + * for the renderer. + * + * Multi-section patches are preflighted up front via {@link Patcher.prepare} + * so a partial batch never lands; the commit loop then narrows the LSP + * batch's `flush` flag to true only for the final write so diagnostics + * round-trip once. + */ +import { + buildCompactDiffPreview, + MismatchError as HashlineMismatchError, + Patch, + Patcher, + type PatchSectionResult, + type PreparedSection, +} from "@oh-my-pi/hashline"; +import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { FileDiagnosticsResult, WritethroughCallback, WritethroughDeferredHandle } from "../../lsp"; +import type { ToolSession } from "../../tools"; +import { outputMeta } from "../../tools/output-meta"; +import { generateDiffString } from "../diff"; +import { getFileSnapshotStore } from "../file-snapshot-store"; +import type { EditToolDetails, EditToolPerFileResult, LspBatchRequest } from "../renderer"; +import { HashlineFilesystem } from "./filesystem"; +import { type HashlineParams, hashlineEditParamsSchema } from "./params"; + +export interface ExecuteHashlineSingleOptions { + session: ToolSession; + input: string; + path?: string; + signal?: AbortSignal; + batchRequest?: LspBatchRequest; + writethrough: WritethroughCallback; + beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; +} + +function getHashlineApplyOptions(session: ToolSession): { autoDropPureInsertDuplicates: boolean } { + return { + autoDropPureInsertDuplicates: session.settings.get("edit.hashlineAutoDropPureInsertDuplicates"), + }; +} + +function noChangeDiagnostic(path: string): string { + return `Edits to ${path} resulted in no changes being made.`; +} + +function narrowBatchRequest(outer: LspBatchRequest | undefined, isLast: boolean): LspBatchRequest | undefined { + if (!outer) return undefined; + return { id: outer.id, flush: isLast && outer.flush }; +} + +interface RenderedSection { + toolResult: AgentToolResult; + perFileResult: EditToolPerFileResult; +} + +function renderSection(result: PatchSectionResult, diagnostics: FileDiagnosticsResult | undefined): RenderedSection { + if (result.op === "noop") { + const toolResult: AgentToolResult = { + content: [{ type: "text", text: noChangeDiagnostic(result.path) }], + details: { diff: "", op: "update", meta: outputMeta().get() }, + }; + return { + toolResult, + perFileResult: { path: result.path, diff: "", op: "update" }, + }; + } + + const diff = generateDiffString(result.before, result.after); + const preview = buildCompactDiffPreview(diff.diff); + const meta = outputMeta() + .diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? []) + .get(); + + const warningsBlock = result.warnings.length > 0 ? `\n\nWarnings:\n${result.warnings.join("\n")}` : ""; + const previewBlock = preview.preview ? `\n${preview.preview}` : ""; + const headline = preview.preview + ? `${result.path}:` + : result.op === "create" + ? `Created ${result.path}` + : `Updated ${result.path}`; + + const firstChangedLine = result.firstChangedLine ?? diff.firstChangedLine; + return { + toolResult: { + content: [{ type: "text", text: `${headline}\n${result.header}${previewBlock}${warningsBlock}` }], + details: { + diff: diff.diff, + firstChangedLine, + diagnostics, + op: result.op, + meta, + }, + }, + perFileResult: { + path: result.path, + diff: diff.diff, + firstChangedLine, + diagnostics, + op: result.op, + }, + }; +} + +export async function executeHashlineSingle( + options: ExecuteHashlineSingleOptions, +): Promise> { + const patch = Patch.parse(options.input, { cwd: options.session.cwd, path: options.path }); + if (patch.sections.length === 0) { + throw new Error("No hashline sections found in input."); + } + + const fs = new HashlineFilesystem({ + session: options.session, + writethrough: options.writethrough, + beginDeferredDiagnosticsForPath: options.beginDeferredDiagnosticsForPath, + signal: options.signal, + batchRequest: options.batchRequest, + }); + const snapshots = getFileSnapshotStore(options.session); + const applyOptions = getHashlineApplyOptions(options.session); + const patcher = new Patcher({ fs, snapshots, applyOptions }); + + // Single-section fast path: prepare, commit, render. + if (patch.sections.length === 1) { + fs.setBatchRequest(narrowBatchRequest(options.batchRequest, true)); + const prepared = await patcher.prepare(patch.sections[0]); + const sectionResult = await patcher.commit(prepared); + if (sectionResult.op === "noop") { + return renderSection(sectionResult, undefined).toolResult; + } + return renderSection(sectionResult, fs.consumeDiagnostics(sectionResult.path)).toolResult; + } + + // Multi-section: prepare every section up front so we fail fast before + // any write hits the filesystem. + const prepared: PreparedSection[] = []; + for (const section of patch.sections) prepared.push(await patcher.prepare(section)); + for (const entry of prepared) { + if (entry.isNoop) throw new Error(noChangeDiagnostic(entry.section.path)); + } + // Then commit each one, narrowing the LSP batch flush flag to the final + // section only. A no-op apply mid-batch is treated as a hard failure — + // the model authored anchors that match the current file content. + const rendered: RenderedSection[] = []; + for (let i = 0; i < prepared.length; i++) { + const isLast = i === prepared.length - 1; + fs.setBatchRequest(narrowBatchRequest(options.batchRequest, isLast)); + const sectionResult = await patcher.commit(prepared[i]); + if (sectionResult.op === "noop") throw new Error(noChangeDiagnostic(sectionResult.path)); + rendered.push(renderSection(sectionResult, fs.consumeDiagnostics(sectionResult.path))); + } + + return { + content: [ + { + type: "text", + text: rendered + .map(r => r.toolResult.content.map(part => (part.type === "text" ? part.text : "")).join("\n")) + .join("\n\n"), + }, + ], + details: { + diff: rendered.map(r => r.toolResult.details?.diff ?? "").join("\n"), + perFileResults: rendered.map(r => r.perFileResult), + }, + }; +} + +export { HashlineMismatchError, type HashlineParams, hashlineEditParamsSchema }; diff --git a/packages/coding-agent/src/edit/hashline/filesystem.ts b/packages/coding-agent/src/edit/hashline/filesystem.ts new file mode 100644 index 000000000..eff843793 --- /dev/null +++ b/packages/coding-agent/src/edit/hashline/filesystem.ts @@ -0,0 +1,125 @@ +/** + * Coding-agent specific {@link Filesystem} adapter for the hashline patcher. + * + * Wires hashline's storage abstraction to the agent runtime: + * + * - Section paths are resolved through the plan-mode redirect so a bare + * `PLAN.md` lands at the canonical session artifact location. + * - Reads go through `readEditFileText` (notebook-aware) and the + * auto-generated-file guard. + * - Writes go through `serializeEditFileText` (notebook-aware) and the + * LSP writethrough, with FS-scan cache invalidation on success. The + * resulting `FileDiagnosticsResult` is captured per-path so the + * orchestrator can attach it to the tool result. + * + * Construct one per `executeHashlineSingle` call: per-section state + * (batch request, diagnostics) lives on the instance and isn't safe to + * share across concurrent edit tools. + */ +import { Filesystem, NotFoundError, type WriteResult } from "@oh-my-pi/hashline"; +import { isEnoent } from "@oh-my-pi/pi-utils"; +import type { FileDiagnosticsResult, WritethroughCallback, WritethroughDeferredHandle } from "../../lsp"; +import type { ToolSession } from "../../tools"; +import { assertEditableFileContent } from "../../tools/auto-generated-guard"; +import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation"; +import { enforcePlanModeWrite, resolvePlanPath } from "../../tools/plan-mode-guard"; +import { readEditFileText, serializeEditFileText } from "../read-file"; +import type { LspBatchRequest } from "../renderer"; + +export interface HashlineFilesystemOptions { + session: ToolSession; + writethrough: WritethroughCallback; + beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; + signal?: AbortSignal; + /** + * Outer LSP batch request inherited from the tool-call context. The + * orchestrator narrows this per-section (flush only on the final write) + * via {@link HashlineFilesystem.setBatchRequest}. + */ + batchRequest?: LspBatchRequest; +} + +export class HashlineFilesystem extends Filesystem { + readonly session: ToolSession; + readonly #writethrough: WritethroughCallback; + readonly #beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; + readonly #signal: AbortSignal | undefined; + #batchRequest: LspBatchRequest | undefined; + #diagnosticsByPath = new Map(); + + constructor(options: HashlineFilesystemOptions) { + super(); + this.session = options.session; + this.#writethrough = options.writethrough; + this.#beginDeferredDiagnosticsForPath = options.beginDeferredDiagnosticsForPath; + this.#signal = options.signal; + this.#batchRequest = options.batchRequest; + } + + /** + * Set the LSP batch request used for the next {@link writeText} call. + * Multi-section orchestrators flip the `flush` flag to true before the + * final section so LSP diagnostics flush in one round-trip. + */ + setBatchRequest(batchRequest: LspBatchRequest | undefined): void { + this.#batchRequest = batchRequest; + } + + /** + * Look up (and clear) the diagnostics captured by the most-recent + * {@link writeText} call for `path`. Returns `undefined` if no write + * has happened or the writethrough returned no diagnostics. + */ + consumeDiagnostics(path: string): FileDiagnosticsResult | undefined { + const value = this.#diagnosticsByPath.get(path); + this.#diagnosticsByPath.delete(path); + return value; + } + + resolveAbsolute(relativePath: string): string { + return resolvePlanPath(this.session, relativePath); + } + + canonicalPath(relativePath: string): string { + return this.resolveAbsolute(relativePath); + } + + async readText(relativePath: string): Promise { + const absolutePath = this.resolveAbsolute(relativePath); + let content: string; + try { + content = await readEditFileText(absolutePath, relativePath); + } catch (error) { + if (isEnoent(error)) throw new NotFoundError(relativePath, error); + if (error instanceof Error && error.message === `File not found: ${relativePath}`) { + throw new NotFoundError(relativePath, error); + } + throw error; + } + // Refuse edits against generated files (lockfiles, models.json, …). + assertEditableFileContent(content, relativePath); + return content; + } + + async writeText(relativePath: string, content: string): Promise { + enforcePlanModeWrite(this.session, relativePath, { op: "update" }); + const absolutePath = this.resolveAbsolute(relativePath); + const finalContent = await serializeEditFileText(absolutePath, relativePath, content); + const diagnostics = await this.#writethrough( + absolutePath, + finalContent, + this.#signal, + Bun.file(absolutePath), + this.#batchRequest, + dst => (dst === absolutePath ? this.#beginDeferredDiagnosticsForPath(absolutePath) : undefined), + ); + invalidateFsScanAfterWrite(absolutePath); + this.#diagnosticsByPath.set(relativePath, diagnostics); + return { text: finalContent }; + } + + async exists(relativePath: string): Promise { + const absolutePath = this.resolveAbsolute(relativePath); + return Bun.file(absolutePath).exists(); + } +} diff --git a/packages/coding-agent/src/edit/hashline/index.ts b/packages/coding-agent/src/edit/hashline/index.ts new file mode 100644 index 000000000..7cc3f4569 --- /dev/null +++ b/packages/coding-agent/src/edit/hashline/index.ts @@ -0,0 +1,4 @@ +export * from "./diff"; +export * from "./execute"; +export * from "./filesystem"; +export * from "./params"; diff --git a/packages/coding-agent/src/edit/hashline/params.ts b/packages/coding-agent/src/edit/hashline/params.ts new file mode 100644 index 000000000..53f462706 --- /dev/null +++ b/packages/coding-agent/src/edit/hashline/params.ts @@ -0,0 +1,11 @@ +/** + * Zod schema for the `edit` tool's hashline mode payload. The schema is + * deliberately permissive (`.passthrough()`) so providers can attach extra + * keys without rejection; only `input` is required and `path` is an + * optional fallback used when the input lacks a `¶PATH#HASH` header. + */ +import * as z from "zod/v4"; + +export const hashlineEditParamsSchema = z.object({ input: z.string(), path: z.string().optional() }).passthrough(); + +export type HashlineParams = z.infer; diff --git a/packages/coding-agent/src/edit/index.ts b/packages/coding-agent/src/edit/index.ts index b82d720d5..e688f91de 100644 --- a/packages/coding-agent/src/edit/index.ts +++ b/packages/coding-agent/src/edit/index.ts @@ -1,13 +1,8 @@ +import { MismatchError as HashlineMismatchError } from "@oh-my-pi/hashline"; +import hashlineGrammar from "@oh-my-pi/hashline/grammar.lark" with { type: "text" }; +import hashlineDescription from "@oh-my-pi/hashline/prompt.md" with { type: "text" }; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import { prompt } from "@oh-my-pi/pi-utils"; -import { - executeHashlineSingle, - HashlineMismatchError, - type HashlineParams, - hashlineEditParamsSchema, -} from "../hashline"; -import hashlineGrammarTemplate from "../hashline/grammar.lark" with { type: "text" }; -import { resolveHashlineGrammarPlaceholders } from "../hashline/hash"; import { createLspWritethrough, type FileDiagnosticsResult, @@ -16,28 +11,25 @@ import { writethroughNoop, } from "../lsp"; import applyPatchDescription from "../prompts/tools/apply-patch.md" with { type: "text" }; -import hashlineDescription from "../prompts/tools/hashline.md" with { type: "text" }; import patchDescription from "../prompts/tools/patch.md" with { type: "text" }; import replaceDescription from "../prompts/tools/replace.md" with { type: "text" }; import type { ToolSession } from "../tools"; import { truncateForPrompt } from "../tools/approval"; import { isInternalUrlPath } from "../tools/path-utils"; import { type EditMode, normalizeEditMode, resolveEditMode } from "../utils/edit-mode"; +import { executeHashlineSingle, type HashlineParams, hashlineEditParamsSchema } from "./hashline"; import { type ApplyPatchParams, applyPatchSchema, expandApplyPatchToEntries } from "./modes/apply-patch"; import applyPatchGrammar from "./modes/apply-patch.lark" with { type: "text" }; import { executePatchSingle, type PatchEditEntry, type PatchParams, patchEditSchema } from "./modes/patch"; import { executeReplaceSingle, type ReplaceEditEntry, type ReplaceParams, replaceEditSchema } from "./modes/replace"; import { type EditToolDetails, type EditToolPerFileResult, getLspBatchRequest, type LspBatchRequest } from "./renderer"; +export * from "@oh-my-pi/hashline"; export { DEFAULT_EDIT_MODE, type EditMode, normalizeEditMode } from "../utils/edit-mode"; export * from "./apply-patch"; export * from "./diff"; -export * from "./file-read-cache"; - -// Resolve hashline grammar placeholders from the TypeScript constants. -const hashlineGrammar = resolveHashlineGrammarPlaceholders(hashlineGrammarTemplate); - -export * from "../hashline"; +export * from "./file-snapshot-store"; +export * from "./hashline"; export * from "./modes/apply-patch"; export * from "./modes/patch"; export * from "./modes/replace"; diff --git a/packages/coding-agent/src/edit/normalize.ts b/packages/coding-agent/src/edit/normalize.ts index 757dde5a9..fc7f16e87 100644 --- a/packages/coding-agent/src/edit/normalize.ts +++ b/packages/coding-agent/src/edit/normalize.ts @@ -1,51 +1,21 @@ /** * Text normalization utilities for the edit tool. * - * Handles line endings, BOM, whitespace, and Unicode normalization. + * Whitespace, Unicode, and indentation helpers. Line-ending and BOM + * primitives live in `@oh-my-pi/hashline` and are re-exported here so + * existing consumers see one stable surface. */ import { padding } from "@oh-my-pi/pi-tui"; -// ═══════════════════════════════════════════════════════════════════════════ -// Line Ending Utilities -// ═══════════════════════════════════════════════════════════════════════════ - -export type LineEnding = "\r\n" | "\n"; - -/** Detect the predominant line ending in content */ -export function detectLineEnding(content: string): LineEnding { - const crlfIdx = content.indexOf("\r\n"); - const lfIdx = content.indexOf("\n"); - if (lfIdx === -1) return "\n"; - if (crlfIdx === -1) return "\n"; - return crlfIdx < lfIdx ? "\r\n" : "\n"; -} - -/** Normalize all line endings to LF */ -export function normalizeToLF(text: string): string { - return text.replace(/\r\n?/g, "\n"); -} - -/** Restore line endings to the specified type */ -export function restoreLineEndings(text: string, ending: LineEnding): string { - return ending === "\r\n" ? text.replace(/\n/g, "\r\n") : text; -} - -// ═══════════════════════════════════════════════════════════════════════════ -// BOM Handling -// ═══════════════════════════════════════════════════════════════════════════ - -export interface BomResult { - /** The BOM character if present, empty string otherwise */ - bom: string; - /** The text without the BOM */ - text: string; -} - -/** Strip UTF-8 BOM if present */ -export function stripBom(content: string): BomResult { - return content.startsWith("\uFEFF") ? { bom: "\uFEFF", text: content.slice(1) } : { bom: "", text: content }; -} +export { + type BomResult, + detectLineEnding, + type LineEnding, + normalizeToLF, + restoreLineEndings, + stripBom, +} from "@oh-my-pi/hashline"; // ═══════════════════════════════════════════════════════════════════════════ // Whitespace Utilities diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index f28c4cdeb..e398b07ca 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -2,11 +2,11 @@ * Edit tool renderer and LSP batching helpers. */ +import { HL_FILE_PREFIX } from "@oh-my-pi/hashline"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text, visibleWidth, wrapTextWithAnsi } from "@oh-my-pi/pi-tui"; import { sanitizeText } from "@oh-my-pi/pi-utils"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { HL_FILE_PREFIX } from "../hashline/hash"; import type { FileDiagnosticsResult } from "../lsp"; import { renderDiff as renderDiffColored } from "../modes/components/diff"; import { getLanguageFromPath, type Theme } from "../modes/theme/theme"; diff --git a/packages/coding-agent/src/edit/streaming.ts b/packages/coding-agent/src/edit/streaming.ts index 9dde960bb..639ca15e1 100644 --- a/packages/coding-agent/src/edit/streaming.ts +++ b/packages/coding-agent/src/edit/streaming.ts @@ -13,22 +13,21 @@ * the injected `editMode` rather than probing argument shape. */ -import { sanitizeText } from "@oh-my-pi/pi-utils"; import { ABORT_MARKER, BEGIN_PATCH_MARKER, - computeHashlineDiff, - computeHashlineSectionDiff, containsRecognizableHashlineOperations, END_PATCH_MARKER, - type HashlineInputSection, - HashlineTokenizer, - splitHashlineInputs, -} from "../hashline"; + type PatchSection as HashlineInputSection, + Patch as HashlinePatch, + Tokenizer as HashlineTokenizer, +} from "@oh-my-pi/hashline"; +import { sanitizeText } from "@oh-my-pi/pi-utils"; import type { Theme } from "../modes/theme/theme"; import { replaceTabs, truncateToWidth } from "../tools/render-utils"; import { type EditMode, resolveEditMode } from "../utils/edit-mode"; import { computeEditDiff, type DiffError, type DiffResult } from "./diff"; +import { computeHashlineDiff, computeHashlineSectionDiff } from "./hashline/diff"; import { type ApplyPatchEntry, expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch"; import { computePatchDiff, type PatchEditEntry } from "./modes/patch"; import type { ReplaceEditEntry } from "./modes/replace"; @@ -438,9 +437,9 @@ const hashlineStrategy: EditStreamingStrategy = { } ctx.signal.throwIfAborted(); - let sections: HashlineInputSection[]; + let sections: readonly HashlineInputSection[]; try { - sections = splitHashlineInputs(input, { cwd: ctx.cwd, path: args.path }); + sections = HashlinePatch.parse(input, { cwd: ctx.cwd, path: args.path }).sections; } catch { // Single-section fallback keeps the original error rendering for the // "haven't typed `¶ PATH` yet" case. diff --git a/packages/coding-agent/src/hashline/bigrams.json b/packages/coding-agent/src/hashline/bigrams.json deleted file mode 100644 index 113f295c7..000000000 --- a/packages/coding-agent/src/hashline/bigrams.json +++ /dev/null @@ -1,649 +0,0 @@ -[ - "aa", - "ab", - "ac", - "ad", - "ae", - "af", - "ag", - "ah", - "ai", - "aj", - "ak", - "al", - "am", - "an", - "ao", - "ap", - "aq", - "ar", - "as", - "at", - "au", - "av", - "aw", - "ax", - "ay", - "az", - "ba", - "bb", - "bc", - "bd", - "be", - "bf", - "bg", - "bh", - "bi", - "bj", - "bk", - "bl", - "bm", - "bn", - "bo", - "bp", - "br", - "bs", - "bt", - "bu", - "bv", - "bw", - "bx", - "by", - "bz", - "ca", - "cb", - "cc", - "cd", - "ce", - "cf", - "cg", - "ch", - "ci", - "cj", - "ck", - "cl", - "cm", - "cn", - "co", - "cp", - "cq", - "cr", - "cs", - "ct", - "cu", - "cv", - "cw", - "cx", - "cy", - "cz", - "da", - "db", - "dc", - "dd", - "de", - "df", - "dg", - "dh", - "di", - "dj", - "dk", - "dl", - "dm", - "dn", - "do", - "dp", - "dq", - "dr", - "ds", - "dt", - "du", - "dv", - "dw", - "dx", - "dy", - "dz", - "ea", - "eb", - "ec", - "ed", - "ee", - "ef", - "eg", - "eh", - "ei", - "ej", - "ek", - "el", - "em", - "en", - "eo", - "ep", - "eq", - "er", - "es", - "et", - "eu", - "ev", - "ew", - "ex", - "ey", - "ez", - "fa", - "fb", - "fc", - "fd", - "fe", - "ff", - "fg", - "fh", - "fi", - "fj", - "fk", - "fl", - "fm", - "fn", - "fo", - "fp", - "fq", - "fr", - "fs", - "ft", - "fu", - "fv", - "fw", - "fx", - "fy", - "fz", - "ga", - "gb", - "gc", - "gd", - "ge", - "gf", - "gg", - "gh", - "gi", - "gj", - "gl", - "gm", - "gn", - "go", - "gp", - "gr", - "gs", - "gt", - "gu", - "gv", - "gw", - "gx", - "gy", - "gz", - "ha", - "hb", - "hc", - "hd", - "he", - "hf", - "hg", - "hh", - "hi", - "hj", - "hk", - "hl", - "hm", - "hn", - "ho", - "hp", - "hq", - "hr", - "hs", - "ht", - "hu", - "hv", - "hw", - "hx", - "hy", - "hz", - "ia", - "ib", - "ic", - "id", - "ie", - "if", - "ig", - "ih", - "ii", - "ij", - "ik", - "il", - "im", - "in", - "io", - "ip", - "iq", - "ir", - "is", - "it", - "iu", - "iv", - "iw", - "ix", - "iy", - "iz", - "ja", - "jb", - "jc", - "jd", - "je", - "jf", - "jg", - "jh", - "ji", - "jj", - "jk", - "jl", - "jm", - "jn", - "jo", - "jp", - "jq", - "jr", - "js", - "jt", - "ju", - "jw", - "jx", - "jy", - "ka", - "kb", - "kc", - "kd", - "ke", - "kf", - "kg", - "kh", - "ki", - "kj", - "kk", - "kl", - "km", - "kn", - "ko", - "kp", - "kr", - "ks", - "kt", - "ku", - "kv", - "kw", - "kx", - "ky", - "la", - "lb", - "lc", - "ld", - "le", - "lf", - "lg", - "lh", - "li", - "lj", - "lk", - "ll", - "lm", - "ln", - "lo", - "lp", - "lr", - "ls", - "lt", - "lu", - "lv", - "lw", - "lx", - "ly", - "lz", - "ma", - "mb", - "mc", - "md", - "me", - "mf", - "mg", - "mh", - "mi", - "mj", - "mk", - "ml", - "mm", - "mn", - "mo", - "mp", - "mq", - "mr", - "ms", - "mt", - "mu", - "mv", - "mw", - "mx", - "my", - "mz", - "na", - "nb", - "nc", - "nd", - "ne", - "nf", - "ng", - "nh", - "ni", - "nj", - "nk", - "nl", - "nm", - "nn", - "no", - "np", - "nr", - "ns", - "nt", - "nu", - "nv", - "nw", - "nx", - "ny", - "nz", - "oa", - "ob", - "oc", - "od", - "oe", - "of", - "og", - "oh", - "oi", - "oj", - "ok", - "ol", - "om", - "on", - "oo", - "op", - "oq", - "or", - "os", - "ot", - "ou", - "ov", - "ow", - "ox", - "oy", - "oz", - "pa", - "pb", - "pc", - "pd", - "pe", - "pf", - "pg", - "ph", - "pi", - "pj", - "pk", - "pl", - "pm", - "pn", - "po", - "pp", - "pq", - "pr", - "ps", - "pt", - "pu", - "pv", - "pw", - "px", - "py", - "pz", - "qa", - "qb", - "qc", - "qd", - "qe", - "qh", - "qi", - "ql", - "qm", - "qn", - "qo", - "qp", - "qq", - "qr", - "qs", - "qt", - "qu", - "qw", - "qx", - "qy", - "ra", - "rb", - "rc", - "rd", - "re", - "rf", - "rg", - "rh", - "ri", - "rk", - "rl", - "rm", - "rn", - "ro", - "rp", - "rq", - "rr", - "rs", - "rt", - "ru", - "rv", - "rw", - "rx", - "ry", - "rz", - "sa", - "sb", - "sc", - "sd", - "se", - "sf", - "sg", - "sh", - "si", - "sj", - "sk", - "sl", - "sm", - "sn", - "so", - "sp", - "sq", - "sr", - "ss", - "st", - "su", - "sv", - "sw", - "sx", - "sy", - "sz", - "ta", - "tb", - "tc", - "td", - "te", - "tf", - "tg", - "th", - "ti", - "tj", - "tk", - "tl", - "tm", - "tn", - "to", - "tp", - "tr", - "ts", - "tt", - "tu", - "tv", - "tw", - "tx", - "ty", - "tz", - "ua", - "ub", - "uc", - "ud", - "ue", - "uf", - "ug", - "uh", - "ui", - "uj", - "uk", - "ul", - "um", - "un", - "uo", - "up", - "uq", - "ur", - "us", - "ut", - "uu", - "uv", - "uw", - "ux", - "uy", - "uz", - "va", - "vb", - "vc", - "vd", - "ve", - "vf", - "vg", - "vh", - "vi", - "vj", - "vk", - "vl", - "vm", - "vn", - "vo", - "vp", - "vq", - "vr", - "vs", - "vt", - "vu", - "vv", - "vw", - "vx", - "vy", - "vz", - "wa", - "wb", - "wc", - "wd", - "we", - "wf", - "wg", - "wh", - "wi", - "wj", - "wk", - "wl", - "wm", - "wn", - "wo", - "wp", - "wr", - "ws", - "wt", - "wu", - "wv", - "ww", - "wx", - "wy", - "xa", - "xb", - "xc", - "xd", - "xe", - "xf", - "xh", - "xi", - "xl", - "xm", - "xn", - "xo", - "xp", - "xr", - "xs", - "xt", - "xu", - "xx", - "xy", - "xz", - "ya", - "yb", - "yc", - "yd", - "ye", - "yf", - "yg", - "yh", - "yi", - "yj", - "yk", - "yl", - "ym", - "yn", - "yo", - "yp", - "yr", - "ys", - "yt", - "yu", - "yv", - "yw", - "yx", - "yy", - "yz", - "za", - "zb", - "zc", - "zd", - "ze", - "zf", - "zg", - "zh", - "zi", - "zk", - "zl", - "zm", - "zn", - "zo", - "zp", - "zr", - "zs", - "zt", - "zu", - "zw", - "zx", - "zy", - "zz" -] diff --git a/packages/coding-agent/src/hashline/diff.ts b/packages/coding-agent/src/hashline/diff.ts deleted file mode 100644 index a6e97691d..000000000 --- a/packages/coding-agent/src/hashline/diff.ts +++ /dev/null @@ -1,82 +0,0 @@ -import { generateDiffString } from "../edit/diff"; -import { normalizeToLF, stripBom } from "../edit/normalize"; -import { readEditFileText } from "../edit/read-file"; -import { resolveToCwd } from "../tools/path-utils"; -import { applyHashlineEdits } from "./apply"; -import { parseHashline } from "./executor"; -import { computeFileHash } from "./hash"; -import { splitHashlineInputs } from "./input"; -import type { HashlineApplyOptions, HashlineEdit, HashlineInputSection } from "./types"; - -async function readHashlineFileText( - _file: { text(): Promise }, - absolutePath: string, - pathText: string, -): Promise { - try { - return await readEditFileText(absolutePath, pathText); - } catch (error) { - const message = error instanceof Error ? error.message : String(error); - throw new Error(message || `Unable to read ${pathText}`); - } -} - -function hasAnchorScopedEdit(edits: readonly HashlineEdit[]): boolean { - return edits.some(edit => { - if (edit.kind === "delete") return true; - return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor"; - }); -} - -function validateSectionHash( - section: HashlineInputSection, - text: string, - edits: readonly HashlineEdit[], -): string | null { - if (section.fileHash === undefined) { - return hasAnchorScopedEdit(edits) - ? `Missing hashline file hash for anchored edit to ${section.path}; use \`¶${section.path}#hash\` from your latest read.` - : null; - } - const currentHash = computeFileHash(text); - if (currentHash === section.fileHash) return null; - return `Hashline file hash mismatch for ${section.path}: section is bound to #${section.fileHash}, but current file hashes to #${currentHash}; re-read and try again.`; -} - -export async function computeHashlineSectionDiff( - section: HashlineInputSection, - cwd: string, - options: HashlineApplyOptions = {}, -): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> { - try { - const absolutePath = resolveToCwd(section.path, cwd); - const rawContent = await readHashlineFileText(Bun.file(absolutePath), absolutePath, section.path); - const { text: content } = stripBom(rawContent); - const normalized = normalizeToLF(content); - const { edits } = parseHashline(section.diff); - const hashError = validateSectionHash(section, normalized, edits); - if (hashError) return { error: hashError }; - const result = applyHashlineEdits(normalized, edits, options); - if (normalized === result.lines) return { error: `No changes would be made to ${section.path}.` }; - return generateDiffString(normalized, result.lines); - } catch (err) { - return { error: err instanceof Error ? err.message : String(err) }; - } -} - -export async function computeHashlineDiff( - input: { input: string; path?: string }, - cwd: string, - options: HashlineApplyOptions = {}, -): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> { - let sections: HashlineInputSection[]; - try { - sections = splitHashlineInputs(input.input, { cwd, path: input.path }); - } catch (err) { - return { error: err instanceof Error ? err.message : String(err) }; - } - if (sections.length !== 1) { - return { error: "Streaming diff preview supports exactly one hashline section." }; - } - return computeHashlineSectionDiff(sections[0], cwd, options); -} diff --git a/packages/coding-agent/src/hashline/execute.ts b/packages/coding-agent/src/hashline/execute.ts deleted file mode 100644 index b50e37a28..000000000 --- a/packages/coding-agent/src/hashline/execute.ts +++ /dev/null @@ -1,334 +0,0 @@ -import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; -import { isEnoent } from "@oh-my-pi/pi-utils"; -import { generateDiffString } from "../edit/diff"; -import { getFileReadCache } from "../edit/file-read-cache"; -import { detectLineEnding, normalizeToLF, restoreLineEndings, stripBom } from "../edit/normalize"; -import { readEditFileText, serializeEditFileText } from "../edit/read-file"; -import type { EditToolDetails } from "../edit/renderer"; -import type { ToolSession } from "../tools"; -import { assertEditableFileContent } from "../tools/auto-generated-guard"; -import { invalidateFsScanAfterWrite } from "../tools/fs-cache-invalidation"; -import { outputMeta } from "../tools/output-meta"; -import { enforcePlanModeWrite, resolvePlanPath } from "../tools/plan-mode-guard"; -import { HashlineMismatchError } from "./anchors"; -import { applyHashlineEdits, type HashlineApplyResult } from "./apply"; -import { buildCompactHashlineDiffPreview } from "./diff-preview"; -import { parseHashline } from "./executor"; -import { computeFileHash, formatHashlineHeader } from "./hash"; -import { splitHashlineInputs } from "./input"; -import { tryRecoverHashlineWithCache } from "./recovery"; -import type { - ExecuteHashlineSingleOptions, - HashlineApplyOptions, - HashlineEdit, - HashlineInputSection, - hashlineEditParamsSchema, -} from "./types"; - -interface ReadHashlineFileResult { - exists: boolean; - rawContent: string; -} - -async function readHashlineFile(absolutePath: string, pathText: string): Promise { - try { - return { exists: true, rawContent: await readEditFileText(absolutePath, pathText) }; - } catch (error) { - if (isEnoent(error)) return { exists: false, rawContent: "" }; - if (error instanceof Error && error.message === `File not found: ${pathText}`) - return { exists: false, rawContent: "" }; - throw error; - } -} - -function hasAnchorScopedEdit(edits: HashlineEdit[]): boolean { - return edits.some(edit => { - if (edit.kind === "delete") return true; - return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor"; - }); -} - -function collectAnchorLines(edits: HashlineEdit[]): number[] { - const lines = new Set(); - for (const edit of edits) { - if (edit.kind === "delete") { - lines.add(edit.anchor.line); - continue; - } - if (edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor") { - lines.add(edit.cursor.anchor.line); - } - } - return [...lines].sort((a, b) => a - b); -} - -function assertSectionHashAllowed(sectionPath: string, fileHash: string | undefined, edits: HashlineEdit[]): void { - if (fileHash !== undefined || !hasAnchorScopedEdit(edits)) return; - throw new Error( - `Missing hashline file hash for anchored edit to ${sectionPath}; use \`¶${sectionPath}#hash\` from your latest read.`, - ); -} - -function formatNoChangeDiagnostic(pathText: string): string { - return `Edits to ${pathText} resulted in no changes being made.`; -} - -function getHashlineApplyOptions(session: ToolSession): HashlineApplyOptions { - return { - autoDropPureInsertDuplicates: session.settings.get("edit.hashlineAutoDropPureInsertDuplicates"), - }; -} - -function getTextContent(result: AgentToolResult): string { - return result.content.map(part => (part.type === "text" ? part.text : "")).join("\n"); -} - -function getEditDetails(result: AgentToolResult): EditToolDetails { - return result.details ?? { diff: "" }; -} - -/** - * Apply hashline edits with file-hash stale recovery. The section hash gates - * line-number edits against the version shown to the model; if the live file - * drifted, snapshot recovery attempts a strict 3-way merge. - */ -function applyHashlineEditsWithRecovery( - session: ToolSession, - absolutePath: string, - pathText: string, - text: string, - fileHash: string | undefined, - edits: HashlineEdit[], - options: HashlineApplyOptions, -): HashlineApplyResult { - if (fileHash === undefined) return applyHashlineEdits(text, edits, options); - - const currentHash = computeFileHash(text); - if (currentHash === fileHash) return applyHashlineEdits(text, edits, options); - - const cache = getFileReadCache(session); - const recovered = tryRecoverHashlineWithCache({ - cache, - absolutePath, - currentText: text, - fileHash, - edits, - options, - }); - if (recovered) { - return { - lines: recovered.lines, - firstChangedLine: recovered.firstChangedLine, - warnings: recovered.warnings, - }; - } - - throw new HashlineMismatchError({ - path: pathText, - expectedFileHash: fileHash, - actualFileHash: currentHash, - fileLines: text.split("\n"), - anchorLines: collectAnchorLines(edits), - }); -} - -/** - * Run all the front-end checks (notebook guard, parse, plan-mode check, file - * load, edit application) without writing. Used to fail fast before applying - * any changes in a multi-section batch. - */ -async function preflightHashlineSection(options: ExecuteHashlineSingleOptions & HashlineInputSection): Promise { - const { session, path: sectionPath, fileHash, diff } = options; - - const absolutePath = resolvePlanPath(session, sectionPath); - const { edits } = parseHashline(diff); - assertSectionHashAllowed(sectionPath, fileHash, edits); - enforcePlanModeWrite(session, sectionPath, { op: "update" }); - - const source = await readHashlineFile(absolutePath, sectionPath); - if (!source.exists && hasAnchorScopedEdit(edits)) throw new Error(`File not found: ${sectionPath}`); - if (source.exists) assertEditableFileContent(source.rawContent, sectionPath); - - const { text } = stripBom(source.rawContent); - const normalized = normalizeToLF(text); - const result = applyHashlineEditsWithRecovery( - session, - absolutePath, - sectionPath, - normalized, - source.exists ? fileHash : undefined, - edits, - getHashlineApplyOptions(session), - ); - if (normalized === result.lines) throw new Error(formatNoChangeDiagnostic(sectionPath)); -} - -async function executeHashlineSection( - options: ExecuteHashlineSingleOptions & HashlineInputSection, -): Promise> { - const { - session, - path: sourcePath, - fileHash, - diff, - signal, - batchRequest, - writethrough, - beginDeferredDiagnosticsForPath, - } = options; - - const absolutePath = resolvePlanPath(session, sourcePath); - const { edits, warnings: parseWarnings } = parseHashline(diff); - assertSectionHashAllowed(sourcePath, fileHash, edits); - enforcePlanModeWrite(session, sourcePath, { op: "update" }); - - const source = await readHashlineFile(absolutePath, sourcePath); - if (!source.exists && hasAnchorScopedEdit(edits)) throw new Error(`File not found: ${sourcePath}`); - if (source.exists) assertEditableFileContent(source.rawContent, sourcePath); - - const { bom, text } = stripBom(source.rawContent); - const originalEnding = detectLineEnding(text); - const originalNormalized = normalizeToLF(text); - const result = applyHashlineEditsWithRecovery( - session, - absolutePath, - sourcePath, - originalNormalized, - source.exists ? fileHash : undefined, - edits, - getHashlineApplyOptions(session), - ); - - if (originalNormalized === result.lines) { - return { - content: [{ type: "text", text: formatNoChangeDiagnostic(sourcePath) }], - details: { diff: "", op: "update", meta: outputMeta().get() }, - }; - } - - const finalContent = await serializeEditFileText( - absolutePath, - sourcePath, - bom + restoreLineEndings(result.lines, originalEnding), - ); - const diagnostics = await writethrough( - absolutePath, - finalContent, - signal, - Bun.file(absolutePath), - batchRequest, - dst => (dst === absolutePath ? beginDeferredDiagnosticsForPath(absolutePath) : undefined), - ); - invalidateFsScanAfterWrite(absolutePath); - // The post-edit content is the freshest, most authoritative "model view" - // of the file: the model just received it back as the diff/preview. Cache - // it so a follow-up edit anchored against this state can still recover - // if the file is touched out-of-band before the next edit lands. - const newFileHash = computeFileHash(result.lines); - getFileReadCache(session).recordContiguous(absolutePath, 1, result.lines.split("\n"), { - fullText: result.lines, - fileHash: newFileHash, - }); - - const diffResult = generateDiffString(originalNormalized, result.lines); - const meta = outputMeta() - .diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? []) - .get(); - const preview = buildCompactHashlineDiffPreview(diffResult.diff); - - const warnings = [...parseWarnings, ...(result.warnings ?? [])]; - const warningsBlock = warnings.length > 0 ? `\n\nWarnings:\n${warnings.join("\n")}` : ""; - const previewBlock = preview.preview ? `\n${preview.preview}` : ""; - const newHashLine = `\n${formatHashlineHeader(sourcePath, newFileHash)}`; - const headline = preview.preview - ? `${sourcePath}:` - : source.exists - ? `Updated ${sourcePath}` - : `Created ${sourcePath}`; - - return { - content: [{ type: "text", text: `${headline}${newHashLine}${previewBlock}${warningsBlock}` }], - details: { - diff: diffResult.diff, - firstChangedLine: result.firstChangedLine ?? diffResult.firstChangedLine, - diagnostics, - op: source.exists ? "update" : "create", - meta, - }, - }; -} - -export async function executeHashlineSingle( - options: ExecuteHashlineSingleOptions, -): Promise> { - const sections = mergeSamePathSections( - splitHashlineInputs(options.input, { cwd: options.session.cwd, path: options.path }), - ); - - // Fast path: a single section needs no preflight pass. - if (sections.length === 1) return executeHashlineSection({ ...options, ...sections[0] }); - - // Multi-section: validate everything up front so we don't apply a partial batch. - for (const section of sections) await preflightHashlineSection({ ...options, ...section }); - - const results = []; - for (const section of sections) { - results.push({ path: section.path, result: await executeHashlineSection({ ...options, ...section }) }); - } - - return { - content: [{ type: "text", text: results.map(({ result }) => getTextContent(result)).join("\n\n") }], - details: { - diff: results.map(({ result }) => getEditDetails(result).diff).join("\n"), - perFileResults: results.map(({ path: resultPath, result }) => { - const details = getEditDetails(result); - return { - path: resultPath, - diff: details.diff, - firstChangedLine: details.firstChangedLine, - diagnostics: details.diagnostics, - op: details.op, - move: details.move, - meta: details.meta, - }; - }), - }, - }; -} - -/** - * Collapse consecutive or interleaved sections targeting the same path into a - * single section with concatenated diffs. Anchors authored against the same - * file snapshot must be applied as one batch; otherwise the first sub-edit - * shifts line numbers out from under the second's anchors and validation fails. - * Path order is preserved by first occurrence. - */ -function mergeSamePathSections(sections: HashlineInputSection[]): HashlineInputSection[] { - const byPath = new Map(); - for (const section of sections) { - const existing = byPath.get(section.path); - if (existing) { - if ( - existing.fileHash !== undefined && - section.fileHash !== undefined && - existing.fileHash !== section.fileHash - ) { - throw new Error( - `Conflicting hashline file hashes for ${section.path}: #${existing.fileHash} and #${section.fileHash}. Re-read the file and retry with one current header.`, - ); - } - if (existing.fileHash === undefined && section.fileHash !== undefined) existing.fileHash = section.fileHash; - existing.diffs.push(section.diff); - continue; - } - byPath.set(section.path, { - ...(section.fileHash !== undefined ? { fileHash: section.fileHash } : {}), - diffs: [section.diff], - }); - } - return Array.from(byPath, ([path, entry]) => ({ - path, - ...(entry.fileHash !== undefined ? { fileHash: entry.fileHash } : {}), - diff: entry.diffs.join("\n"), - })); -} diff --git a/packages/coding-agent/src/hashline/input.ts b/packages/coding-agent/src/hashline/input.ts deleted file mode 100644 index 7f71a6beb..000000000 --- a/packages/coding-agent/src/hashline/input.ts +++ /dev/null @@ -1,137 +0,0 @@ -import * as path from "node:path"; -import { HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./hash"; -import { HashlineTokenizer } from "./tokenizer"; -import type { HashlineInputSection, SplitHashlineOptions } from "./types"; - -// Pure classification — single shared tokenizer is safe. -const TOKENIZER = new HashlineTokenizer(); - -function unquoteHashlinePath(pathText: string): string { - if (pathText.length < 2) return pathText; - const first = pathText[0]; - const last = pathText[pathText.length - 1]; - if ((first === '"' || first === "'") && first === last) return pathText.slice(1, -1); - return pathText; -} - -function normalizeHashlinePath(rawPath: string, cwd?: string): string { - const unquoted = unquoteHashlinePath(rawPath.trim()); - if (!cwd || !path.isAbsolute(unquoted)) return unquoted; - const relative = path.relative(path.resolve(cwd), path.resolve(unquoted)); - const isWithinCwd = relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative)); - return isWithinCwd ? relative || "." : unquoted; -} - -/** - * Parse a `¶PATH[#hash]` header line. Returns `null` for lines that do not - * begin with the `¶` prefix; throws the existing "Input header must be …" - * error when a `¶`-prefixed line fails the strict shape (so malformed paths - * surface immediately instead of being silently re-classified as payload). - */ -function parseHashlineHeaderLine(line: string, cwd?: string): HashlineInputSection | null { - const trimmed = line.trimEnd(); - if (!trimmed.startsWith(HL_FILE_PREFIX)) return null; - - const token = TOKENIZER.tokenize(trimmed); - if (token.kind !== "header") { - throw new Error( - `Input header must be ${HL_FILE_PREFIX}PATH or ${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}HASH with a 4-hex file hash; got ${JSON.stringify(trimmed)}.`, - ); - } - - const parsedPath = normalizeHashlinePath(token.path, cwd); - if (parsedPath.length === 0) { - throw new Error(`Input header "${HL_FILE_PREFIX}" is empty; provide a file path.`); - } - return token.fileHash !== undefined - ? { path: parsedPath, fileHash: token.fileHash, diff: "" } - : { path: parsedPath, diff: "" }; -} - -function stripLeadingBlankLines(input: string): string { - const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input; - const lines = stripped.split("\n"); - while (lines.length > 0) { - const head = lines[0].replace(/\r$/, ""); - if (head.trim().length === 0 || TOKENIZER.tokenize(head).kind === "envelope-begin") { - lines.shift(); - continue; - } - break; - } - return lines.join("\n"); -} - -export function containsRecognizableHashlineOperations(input: string): boolean { - for (const line of input.split(/\r?\n/)) { - if (TOKENIZER.isOp(line)) return true; - } - return false; -} - -function normalizeFallbackInput(input: string, options: SplitHashlineOptions): string { - const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input; - const hasExplicitHeader = stripped - .split(/\r?\n/) - .some(rawLine => parseHashlineHeaderLine(rawLine, options.cwd) !== null); - if (hasExplicitHeader) return input; - - if (!options.path || !containsRecognizableHashlineOperations(input)) return input; - const fallbackPath = normalizeHashlinePath(options.path, options.cwd); - if (fallbackPath.length === 0) return input; - return `${HL_FILE_PREFIX}${fallbackPath}\n${input}`; -} - -export function splitHashlineInput(input: string, options: SplitHashlineOptions = {}): HashlineInputSection { - const [section] = splitHashlineInputs(input, options); - return section; -} - -export function splitHashlineInputs(input: string, options: SplitHashlineOptions = {}): HashlineInputSection[] { - const stripped = stripLeadingBlankLines(normalizeFallbackInput(input, options)); - const lines = stripped.split(/\r?\n/); - const firstLine = lines[0] ?? ""; - - if (parseHashlineHeaderLine(firstLine, options.cwd) === null) { - const preview = JSON.stringify(firstLine.slice(0, 120)); - throw new Error( - `input must begin with "${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}HASH" on the first non-blank line for anchored edits; got: ${preview}. ` + - `Example: "${HL_FILE_PREFIX}src/foo.ts${HL_FILE_HASH_SEP}1a2b" then edit ops.`, - ); - } - - const sections: HashlineInputSection[] = []; - let current: HashlineInputSection | undefined; - let currentLines: string[] = []; - - const flush = () => { - if (!current) return; - const hasOps = currentLines.some(line => line.trim().length > 0); - if (hasOps) sections.push({ ...current, diff: currentLines.join("\n") }); - currentLines = []; - }; - - for (const line of lines) { - const trimmed = line.trimEnd(); - const token = TOKENIZER.tokenize(line); - if (token.kind === "envelope-end" || token.kind === "abort") break; - if (token.kind === "envelope-begin") continue; - - // Route every `¶`-prefixed line through parseHashlineHeaderLine so - // malformed headers still raise the strict "Input header must be …" - // diagnostic (the tokenizer alone would silently classify them as - // payload). - if (trimmed.startsWith(HL_FILE_PREFIX)) { - const header = parseHashlineHeaderLine(line, options.cwd); - if (header !== null) { - flush(); - current = header; - currentLines = []; - continue; - } - } - currentLines.push(line); - } - flush(); - return sections; -} diff --git a/packages/coding-agent/src/hashline/recovery.ts b/packages/coding-agent/src/hashline/recovery.ts deleted file mode 100644 index 30bcfc125..000000000 --- a/packages/coding-agent/src/hashline/recovery.ts +++ /dev/null @@ -1,139 +0,0 @@ -import * as Diff from "diff"; -import { generateDiffString } from "../edit/diff"; -import type { FileReadCache, FileReadSnapshot } from "../edit/file-read-cache"; -import { applyHashlineEdits, type HashlineApplyResult } from "./apply"; -import { computeFileHash } from "./hash"; -import type { HashlineApplyOptions, HashlineEdit } from "./types"; - -export interface HashlineRecoveryArgs { - cache: FileReadCache; - absolutePath: string; - currentText: string; - fileHash: string; - edits: HashlineEdit[]; - options: HashlineApplyOptions; -} - -export interface HashlineRecoveryResult { - lines: string; - firstChangedLine: number | undefined; - warnings: string[]; -} - -// Section hashes are line-precise; never let Diff.applyPatch slide a hunk onto a -// duplicate closer 100+ lines away. If snapshot replay does not align exactly, -// refuse and let the model re-read. -const HASHLINE_RECOVERY_FUZZ_FACTOR = 0; - -const HASHLINE_RECOVERY_EXTERNAL_WARNING = - "Recovered from a stale file hash using a previous read snapshot (file changed externally between read and edit)."; -const HASHLINE_RECOVERY_SESSION_CHAIN_WARNING = - "Recovered from a stale file hash using an earlier in-session snapshot (the file hash advanced after a prior edit in this session)."; -const HASHLINE_RECOVERY_SESSION_REPLAY_WARNING = - "Recovered by replaying your edits onto the current file content — your previous edit in this session changed line(s) you re-targeted with a stale hash. Verify the diff matches your intent before continuing."; - -function applyEditsToSnapshot( - previousText: string, - currentText: string, - edits: HashlineEdit[], - options: HashlineApplyOptions, - recoveryWarning: string, -): HashlineRecoveryResult | null { - let applied: HashlineApplyResult; - try { - applied = applyHashlineEdits(previousText, edits, options); - } catch { - return null; - } - if (applied.lines === previousText) return null; - - const patch = Diff.structuredPatch("file", "file", previousText, applied.lines, "", "", { context: 3 }); - const merged = Diff.applyPatch(currentText, patch, { fuzzFactor: HASHLINE_RECOVERY_FUZZ_FACTOR }); - if (typeof merged !== "string" || merged === currentText) return null; - - const mergedDiff = generateDiffString(currentText, merged); - const hasNetChange = mergedDiff.firstChangedLine !== undefined; - const recoveryWarnings = hasNetChange - ? [recoveryWarning, ...(applied.warnings ?? [])] - : [...(applied.warnings ?? [])]; - - return { - lines: merged, - firstChangedLine: mergedDiff.firstChangedLine ?? applied.firstChangedLine, - warnings: recoveryWarnings, - }; -} - -function replaySessionChainOnCurrent( - previousText: string, - currentText: string, - edits: HashlineEdit[], - options: HashlineApplyOptions, -): HashlineRecoveryResult | null { - // Only safe when no insert/delete shifted line counts in the prior edit - // chain: if total line counts match, every line number in `edits` still - // resolves to the same logical row. - if (previousText.split("\n").length !== currentText.split("\n").length) return null; - let applied: HashlineApplyResult; - try { - applied = applyHashlineEdits(currentText, edits, options); - } catch { - return null; - } - if (applied.lines === currentText) return null; - return { - lines: applied.lines, - firstChangedLine: applied.firstChangedLine, - warnings: [HASHLINE_RECOVERY_SESSION_REPLAY_WARNING, ...(applied.warnings ?? [])], - }; -} - -function buildSparseOverlayText(currentText: string, snapshotLines: ReadonlyMap): string { - const overlaid = currentText.split("\n"); - let maxCachedLine = 0; - for (const lineNum of snapshotLines.keys()) { - if (lineNum > maxCachedLine) maxCachedLine = lineNum; - } - while (overlaid.length < maxCachedLine) overlaid.push(""); - for (const [lineNum, content] of snapshotLines) { - overlaid[lineNum - 1] = content; - } - return overlaid.join("\n"); -} - -function isHeadSnapshot(head: FileReadSnapshot | null, snapshot: FileReadSnapshot): boolean { - return head === snapshot; -} - -function resolveRecoveryWarning(head: FileReadSnapshot | null, snapshot: FileReadSnapshot): string { - return isHeadSnapshot(head, snapshot) ? HASHLINE_RECOVERY_EXTERNAL_WARNING : HASHLINE_RECOVERY_SESSION_CHAIN_WARNING; -} - -/** - * Attempt to recover from a section file-hash mismatch by replaying the edits - * against a cached pre-edit snapshot of the file and 3-way-merging the result - * onto the current on-disk content. Returns `null` when no recovery is possible. - */ -export function tryRecoverHashlineWithCache(args: HashlineRecoveryArgs): HashlineRecoveryResult | null { - const { cache, absolutePath, currentText, fileHash, edits, options } = args; - const head = cache.get(absolutePath); - const snapshot = cache.getByHash(absolutePath, fileHash); - if (!snapshot || snapshot.lines.size === 0) return null; - - const recoveryWarning = resolveRecoveryWarning(head, snapshot); - const isSessionChain = !isHeadSnapshot(head, snapshot); - if (snapshot.fullText !== undefined) { - const merged = applyEditsToSnapshot(snapshot.fullText, currentText, edits, options, recoveryWarning); - if (merged !== null) return merged; - // Session-chain fast-path: prior in-session edit changed the same line(s) - // the model is now re-targeting with the stale hash. When line counts - // match, the edits' line numbers still resolve to the right rows — replay - // onto the current text directly. - if (isSessionChain) return replaySessionChainOnCurrent(snapshot.fullText, currentText, edits, options); - return null; - } - - const overlayText = buildSparseOverlayText(currentText, snapshot.lines); - if (computeFileHash(overlayText) !== fileHash) return null; - return applyEditsToSnapshot(overlayText, currentText, edits, options, recoveryWarning); -} diff --git a/packages/coding-agent/src/hashline/types.ts b/packages/coding-agent/src/hashline/types.ts deleted file mode 100644 index 1bd87f70a..000000000 --- a/packages/coding-agent/src/hashline/types.ts +++ /dev/null @@ -1,66 +0,0 @@ -import * as z from "zod/v4"; -import type { LspBatchRequest } from "../edit/renderer"; -import type { WritethroughCallback, WritethroughDeferredHandle } from "../lsp"; -import type { ToolSession } from "../tools"; - -export type Anchor = { - line: number; -}; - -export type HashlineCursor = - | { kind: "bof" } - | { kind: "eof" } - | { kind: "before_anchor"; anchor: Anchor } - | { kind: "after_anchor"; anchor: Anchor }; - -export type HashlineEdit = - | { kind: "insert"; cursor: HashlineCursor; text: string; lineNum: number; index: number } - | { kind: "delete"; anchor: Anchor; lineNum: number; index: number; oldAssertion?: string }; - -export interface HashlineInputSection { - path: string; - fileHash?: string; - diff: string; -} - -/** `path` is accepted by the edit tool runtime; other extra keys are preserved. */ -export const hashlineEditParamsSchema = z.object({ input: z.string(), path: z.string().optional() }).passthrough(); -export type HashlineParams = z.infer; - -export interface HashlineStreamOptions { - /** First line number to use when formatting (1-indexed). */ - startLine?: number; - /** Maximum formatted lines per yielded chunk (default: 200). */ - maxChunkLines?: number; - /** Maximum UTF-8 bytes per yielded chunk (default: 64 KiB). */ - maxChunkBytes?: number; -} - -export interface CompactHashlineDiffPreview { - preview: string; - addedLines: number; - removedLines: number; -} - -export interface CompactHashlineDiffOptions { - /** Maximum entries kept on each side of an unchanged-context truncation (default: 2). */ - maxUnchangedRun?: number; -} -export interface HashlineApplyOptions { - autoDropPureInsertDuplicates?: boolean; -} - -export interface SplitHashlineOptions { - cwd?: string; - path?: string; -} - -export interface ExecuteHashlineSingleOptions { - session: ToolSession; - input: string; - path?: string; - signal?: AbortSignal; - batchRequest?: LspBatchRequest; - writethrough: WritethroughCallback; - beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; -} diff --git a/packages/coding-agent/src/index.ts b/packages/coding-agent/src/index.ts index ace3d23ce..b17ec967c 100644 --- a/packages/coding-agent/src/index.ts +++ b/packages/coding-agent/src/index.ts @@ -26,7 +26,6 @@ export * from "./extensibility/extensions"; export * from "./extensibility/skills"; // Slash commands export { type FileSlashCommand, loadSlashCommands as discoverSlashCommands } from "./extensibility/slash-commands"; -export * from "./hashline"; export type * from "./lsp"; // Main entry point export * from "./main"; diff --git a/packages/coding-agent/src/tools/ast-edit.ts b/packages/coding-agent/src/tools/ast-edit.ts index 9ebda0bd3..1e7d47b29 100644 --- a/packages/coding-agent/src/tools/ast-edit.ts +++ b/packages/coding-agent/src/tools/ast-edit.ts @@ -1,4 +1,5 @@ import * as path from "node:path"; +import { computeFileHash, formatHashlineHeader } from "@oh-my-pi/hashline"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import { type AstReplaceChange, type AstReplaceFileChange, astEdit } from "@oh-my-pi/pi-natives"; import type { Component } from "@oh-my-pi/pi-tui"; @@ -6,7 +7,6 @@ import { Text } from "@oh-my-pi/pi-tui"; import { $envpos, prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { computeFileHash, formatHashlineHeader } from "../hashline/hash"; import type { Theme } from "../modes/theme/theme"; import astEditDescription from "../prompts/tools/ast-edit.md" with { type: "text" }; import { Ellipsis, fileHyperlink, renderStatusLine, renderTreeList, truncateToWidth } from "../tui"; diff --git a/packages/coding-agent/src/tools/ast-grep.ts b/packages/coding-agent/src/tools/ast-grep.ts index 132945649..d950014b8 100644 --- a/packages/coding-agent/src/tools/ast-grep.ts +++ b/packages/coding-agent/src/tools/ast-grep.ts @@ -1,13 +1,13 @@ import * as path from "node:path"; +import { computeFileHash, formatHashlineHeader } from "@oh-my-pi/hashline"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import { type AstFindMatch, astGrep } from "@oh-my-pi/pi-natives"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { getFileReadCache } from "../edit/file-read-cache"; +import { getFileSnapshotStore } from "../edit/file-snapshot-store"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { computeFileHash, formatHashlineHeader } from "../hashline/hash"; import type { Theme } from "../modes/theme/theme"; import astGrepDescription from "../prompts/tools/ast-grep.md" with { type: "text" }; import { Ellipsis, fileHyperlink, renderStatusLine, renderTreeList, truncateToWidth } from "../tui"; @@ -268,7 +268,7 @@ export class AstGrepTool implements AgentTool 0) { - getFileReadCache(this.session).recordSparse(hashContext.absolutePath, cacheEntries, { + getFileSnapshotStore(this.session).recordSparse(hashContext.absolutePath, cacheEntries, { fileHash: hashContext.fileHash, }); } diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index a5cc255ce..8cd6d8782 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -1,3 +1,4 @@ +import type { InMemorySnapshotStore } from "@oh-my-pi/hashline"; import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core"; import type { ToolChoice } from "@oh-my-pi/pi-ai"; import { $env, $flag, logger } from "@oh-my-pi/pi-utils"; @@ -230,11 +231,11 @@ export interface ToolSession { /** Set or clear active checkpoint state. */ setCheckpointState?: (state: CheckpointState | null) => void; - /** Per-session cache of file contents as last shown to the model by - * `read`/`search`. Used by hashline anchor-stale recovery to reconstruct - * the version the model authored anchors against when the file changed - * out-of-band. Lazily initialized by `getFileReadCache`. */ - fileReadCache?: import("../edit/file-read-cache").FileReadCache; + /** Per-session snapshot store of file contents as last shown to the model + * by `read`/`search`. Used by hashline anchor-stale recovery to + * reconstruct the version the model authored anchors against when the + * file changed out-of-band. Lazily initialized by `getFileSnapshotStore`. */ + fileSnapshotStore?: InMemorySnapshotStore; /** Per-session log of unresolved git merge conflict regions surfaced by * `read`. Each entry gets a stable id N referenced by `write conflict://N` diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 9d5f5bd9f..04af33cca 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -1,6 +1,7 @@ import { Database } from "bun:sqlite"; import * as fs from "node:fs/promises"; import * as path from "node:path"; +import { computeFileHash, formatHashlineHeader, formatNumberedLine, formatNumberedLines } from "@oh-my-pi/hashline"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; import { glob, type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives"; @@ -8,11 +9,10 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { getRemoteDir, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { getFileReadCache } from "../edit/file-read-cache"; +import { getFileSnapshotStore } from "../edit/file-snapshot-store"; import { normalizeToLF } from "../edit/normalize"; import { isNotebookPath, readEditableNotebookText } from "../edit/notebook"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { computeFileHash, formatHashlineHeader, formatNumberedLine, formatNumberedLines } from "../hashline/hash"; import { InternalUrlRouter } from "../internal-urls"; import { parseInternalUrl } from "../internal-urls/parse"; import type { InternalUrl } from "../internal-urls/types"; @@ -147,7 +147,7 @@ function recordHashlineSnapshot( context: HashlineHeaderContext | undefined, ): void { if (!context || !absolutePath || !path.isAbsolute(absolutePath)) return; - getFileReadCache(session).recordContiguous(absolutePath, 1, context.fullText.split("\n"), { + getFileSnapshotStore(session).recordContiguous(absolutePath, 1, context.fullText.split("\n"), { fullText: context.fullText, fileHash: context.fileHash, }); @@ -1059,7 +1059,7 @@ export class ReadTool implements AgentTool { } if (collectedLines.length > 0) { - getFileReadCache(this.session).recordContiguous( + getFileSnapshotStore(this.session).recordContiguous( absolutePath, range.startLine, collectedLines, @@ -1863,7 +1863,7 @@ export class ReadTool implements AgentTool { : undefined; if (collectedLines.length > 0 && !firstLineExceedsLimit) { - getFileReadCache(this.session).recordContiguous( + getFileSnapshotStore(this.session).recordContiguous( absolutePath, startLineDisplay, collectedLines, diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index 73c4085f9..58929fe7d 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -1,15 +1,15 @@ import { mkdtemp, rm, stat, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import * as path from "node:path"; +import { computeFileHash, formatHashlineHeader } from "@oh-my-pi/hashline"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import { type GrepMatch, GrepOutputMode, type GrepResult, grep } from "@oh-my-pi/pi-natives"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { getFileReadCache } from "../edit/file-read-cache"; +import { getFileSnapshotStore } from "../edit/file-snapshot-store"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { computeFileHash, formatHashlineHeader } from "../hashline/hash"; import type { Theme } from "../modes/theme/theme"; import searchDescription from "../prompts/tools/search.md" with { type: "text" }; import { DEFAULT_MAX_COLUMN, type TruncationResult, truncateHead } from "../session/streaming-output"; @@ -57,7 +57,9 @@ const searchSchema = z pattern: z.string().describe("regex pattern"), paths: z .union([searchPathEntrySchema, z.array(searchPathEntrySchema).min(1)]) - .describe("file, directory, glob, internal URL, or array of those to search; append `:` to scope a file to specific line ranges"), + .describe( + "file, directory, glob, internal URL, or array of those to search; append `:` to scope a file to specific line ranges", + ), i: z.boolean().optional().describe("case-insensitive search"), gitignore: z.boolean().optional().describe("respect gitignore"), skip: z @@ -148,11 +150,7 @@ function parsePathSpecs(rawEntries: readonly string[]): SearchPathSpec[] { return specs; } -function mergeRangesInto( - map: Map, - absKey: string, - ranges: readonly LineRange[], -): void { +function mergeRangesInto(map: Map, absKey: string, ranges: readonly LineRange[]): void { // Concat-without-merge is correct: `isLineInRanges` scans linearly, so // duplicates/overlaps only cost a few extra comparisons per match. const existing = map.get(absKey); @@ -352,9 +350,7 @@ export class SearchTool implements AgentTool 0 && hashContext) { - getFileReadCache(this.session).recordSparse(hashContext.absolutePath, cacheEntries, { + getFileSnapshotStore(this.session).recordSparse(hashContext.absolutePath, cacheEntries, { fileHash: hashContext.fileHash, }); } diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 8111755d1..4f474b2af 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -1,12 +1,12 @@ import { Database } from "bun:sqlite"; import * as fs from "node:fs/promises"; import * as path from "node:path"; +import { stripHashlinePrefixes } from "@oh-my-pi/hashline"; import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { isEnoent, isRecord, prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { stripHashlinePrefixes } from "../edit"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { InternalUrlRouter } from "../internal-urls"; import { parseInternalUrl } from "../internal-urls/parse"; @@ -487,7 +487,7 @@ export class WriteTool implements AgentTool { @@ -907,8 +963,8 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { const b = new FileReadCache(); const fakePath = "/tmp/__hashline-cache-isolation__.ts"; a.recordContiguous(fakePath, 1, ["x", "y", "z"]); - expect(a.get(fakePath)).not.toBeNull(); - expect(b.get(fakePath)).toBeNull(); + expect(a.head(fakePath)).not.toBeNull(); + expect(b.head(fakePath)).toBeNull(); }); it("captures the post-edit result so the next edit can recover from anchors against it", async () => { @@ -931,7 +987,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { await executeHashlineSingle(hashlineExecuteOptions(tempDir, firstInput, undefined, session)); const v1Lines = ["alpha", "BETA", "gamma", "delta", "epsilon"]; expect(await Bun.file(filePath).text()).toBe(`${v1Lines.join("\n")}\n`); - const snap = getFileReadCache(session).get(filePath); + const snap = getFileReadCache(session).head(filePath); expect(snap?.lines.get(1)).toBe("alpha"); expect(snap?.lines.get(2)).toBe("BETA"); expect(snap?.lines.get(3)).toBe("gamma"); @@ -994,9 +1050,9 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { fileHash: computeFileHash(version), }); } - expect(cache.get(fakePath)?.fileHash).toBe(computeFileHash("three\n")); - expect(cache.getByHash(fakePath, computeFileHash("one\n"))?.fullText).toBe("one\n"); - expect(cache.getByHash(fakePath, computeFileHash("two\n"))?.fullText).toBe("two\n"); + expect(cache.head(fakePath)?.fileHash).toBe(computeFileHash("three\n")); + expect(cache.byHash(fakePath, computeFileHash("one\n"))?.fullText).toBe("one\n"); + expect(cache.byHash(fakePath, computeFileHash("two\n"))?.fullText).toBe("two\n"); }); it("drops a cached entry when newly recorded lines disagree on overlap", () => { @@ -1011,7 +1067,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { [7, "g"], ]); - const snap = cache.get(fakePath); + const snap = cache.head(fakePath); expect(snap).not.toBeNull(); // Old entries dropped; only the divergent record's entries remain. expect(snap?.lines.has(1)).toBe(false); @@ -1026,10 +1082,10 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { for (let i = 0; i < 32; i++) { cache.recordContiguous(`/tmp/file-${i}.ts`, 1, ["x"]); } - expect(cache.get("/tmp/file-0.ts")).toBeNull(); - expect(cache.get("/tmp/file-1.ts")).toBeNull(); - expect(cache.get("/tmp/file-2.ts")).not.toBeNull(); - expect(cache.get("/tmp/file-31.ts")).not.toBeNull(); + expect(cache.head("/tmp/file-0.ts")).toBeNull(); + expect(cache.head("/tmp/file-1.ts")).toBeNull(); + expect(cache.head("/tmp/file-2.ts")).not.toBeNull(); + expect(cache.head("/tmp/file-31.ts")).not.toBeNull(); }); }); diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md new file mode 100644 index 000000000..42a7a6356 --- /dev/null +++ b/packages/hashline/CHANGELOG.md @@ -0,0 +1,15 @@ +# Changelog + +All notable changes to this package will be documented in this file. + +## [Unreleased] +### Added + +- Added a high-level `Patcher` API with all-or-nothing `apply` and staged `prepare`/`commit` flows for multi-file patch updates +- Added pluggable `Filesystem` and `SnapshotStore` abstractions with built-in `NodeFilesystem`, `InMemoryFilesystem`, and `InMemorySnapshotStore` adapters +- Added patch parsing that consumes `¶PATH#HASH` hunk headers, validates section file hashes, and supports optional patch envelope markers +- Added tolerant input handling that strips read/search prefixes and supports optional `cwd`/fallback-path resolution when parsing patch payloads +- Added automatic line-ending and BOM normalization on read, with original encoding shape restored on write +- Added follow-up helpers `buildCompactDiffPreview` and `streamHashLines` for compact diff previews and chunked streaming of numbered lines +- Added stale-file-hash recovery that replays edits against snapshots and merges results onto current file content when direct hash validation fails +- Initial standalone release. Extracted from `@oh-my-pi/pi-coding-agent` with a diff --git a/packages/hashline/README.md b/packages/hashline/README.md new file mode 100644 index 000000000..92ae8d2fc --- /dev/null +++ b/packages/hashline/README.md @@ -0,0 +1,78 @@ +# @oh-my-pi/hashline + +A compact, line-anchored patch language and applier. + +Hashline is a diff format designed for LLM-driven file edits. It binds every +hunk to a file-content hash so stale anchors are rejected before they corrupt +code, and it abstracts over the filesystem so the same patcher works on disk, +in memory, over the network, or against any custom backend. + +## Quick start + +```ts +import { + Filesystem, + InMemoryFilesystem, + InMemorySnapshotStore, + Patcher, + Patch, +} from "@oh-my-pi/hashline"; + +const fs = new InMemoryFilesystem(); +await fs.writeText( + "hello.ts", + `const greeting = "hi";\nexport { greeting };\n`, +); + +const patcher = new Patcher({ fs }); +const patch = Patch.parse(`¶hello.ts\n1:\n+const greeting = "hello";`); +const result = await patcher.apply(patch); + +console.log(result.sections[0].op); // "update" +console.log(await fs.readText("hello.ts")); +``` + +## Format + +See [`src/prompt.md`](./src/prompt.md) for the user-facing description and +[`src/grammar.lark`](./src/grammar.lark) for the formal grammar. + +Each hunk starts with a `¶PATH#HASH` header. The hash is a 4-hex-character +xxHash32 truncation of the file's LF-normalized content. The hash protects +against stale anchors: if the file changed between the read that produced the +hash and the edit, the patcher refuses (or, with a `SnapshotStore`, tries +session-aware recovery). + +Inside a hunk: + +|Op|Meaning| +|---|---| +|`LINE↑`|Insert before LINE (or `BOF↑` for the beginning of file)| +|`LINE↓`|Insert after LINE (or `EOF↓` for the end of file)| +|`A-B:`|Replace lines A..B (single-anchor `A:` is sugar for `A-A:`)| +|`A-B!`|Delete lines A..B (single-anchor `A!` is sugar for `A-A!`)| +|`+TEXT`|Payload continuation. The `+` prefix is stripped| + +## Abstractions + +### `Filesystem` + +Read and write text by path. The default implementations: + +- `InMemoryFilesystem` — backed by a `Map`. Tests, sandboxes. +- `NodeFilesystem` — disk-backed via `Bun.file`/`Bun.write`. Default for CLIs. + +Subclass `Filesystem` to wire hashline into any storage: VFS, S3, an LSP +text-document protocol, a Git tree, anything. + +### `SnapshotStore` + +Optional. When provided to `Patcher`, hashline tries to recover from a stale +section hash by replaying the edit against a cached pre-edit snapshot of the +file and 3-way-merging onto the current content. See `recovery.ts`. + +### `Patcher` + +The orchestration class. Reads, normalizes line endings + BOM, applies edits, +restores line endings, and writes via the configured `Filesystem`. Multi-section +patches are preflighted up front so a partial batch never lands. diff --git a/packages/hashline/package.json b/packages/hashline/package.json new file mode 100644 index 000000000..5ef5c2baf --- /dev/null +++ b/packages/hashline/package.json @@ -0,0 +1,61 @@ +{ + "type": "module", + "name": "@oh-my-pi/hashline", + "version": "15.5.3", + "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", + "homepage": "https://omp.sh", + "author": "Can Boluk", + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/can1357/oh-my-pi.git", + "directory": "packages/hashline" + }, + "bugs": { + "url": "https://github.com/can1357/oh-my-pi/issues" + }, + "keywords": [ + "patch", + "diff", + "edit", + "hashline", + "agent", + "llm" + ], + "main": "./src/index.ts", + "types": "./src/index.ts", + "scripts": { + "check": "biome check . && bun run check:types", + "check:types": "tsgo -p tsconfig.json --noEmit", + "lint": "biome lint .", + "test": "bun test", + "fix": "biome check --write --unsafe .", + "fmt": "biome format --write ." + }, + "dependencies": { + "diff": "catalog:", + "lru-cache": "catalog:" + }, + "devDependencies": { + "@types/bun": "catalog:" + }, + "engines": { + "bun": ">=1.3.14" + }, + "files": [ + "src" + ], + "exports": { + ".": { + "types": "./src/index.ts", + "import": "./src/index.ts" + }, + "./grammar.lark": "./src/grammar.lark", + "./prompt.md": "./src/prompt.md", + "./*": { + "types": "./src/*.ts", + "import": "./src/*.ts" + }, + "./*.js": "./src/*.ts" + } +} diff --git a/packages/coding-agent/src/hashline/apply.ts b/packages/hashline/src/apply.ts similarity index 87% rename from packages/coding-agent/src/hashline/apply.ts rename to packages/hashline/src/apply.ts index ececc4544..e4323cd49 100644 --- a/packages/coding-agent/src/hashline/apply.ts +++ b/packages/hashline/src/apply.ts @@ -1,38 +1,41 @@ +/** + * Apply a parsed list of {@link Edit}s to a text body and return the + * post-edit lines plus any diagnostic warnings. Pure function: no FS, no + * mutation of the input. + * + * The applier is conservative about edits that look like authoring mistakes: + * + * - Replace ops on a blank line with non-empty payload are rejected outright + * (the model almost certainly miscounted; recommend `↑`/`↓` instead). + * - Multi-line replacement-boundary duplicates are auto-absorbed (model + * echoed surrounding context as if it were payload). + * - Single-line structural-boundary duplicates (`}`, `)`, `];`, …) are + * auto-absorbed when delimiter balance suggests the range truncated short. + * + * Diagnostics are returned as `warnings[]` in {@link ApplyResult}; they do + * not abort the apply. + */ import { cloneCursor } from "./tokenizer"; -import type { Anchor, HashlineApplyOptions, HashlineCursor, HashlineEdit } from "./types"; +import type { Anchor, ApplyOptions, ApplyResult, Cursor, Edit } from "./types"; -export interface HashlineApplyResult { - lines: string; - firstChangedLine?: number; - warnings?: string[]; - noopEdits?: HashlineNoopEdit[]; -} - -export interface HashlineNoopEdit { - editIndex: number; - loc: string; - reason: string; - current: string; -} - -type HashlineLineOrigin = "original" | "insert" | "replacement"; +type LineOrigin = "original" | "insert" | "replacement"; interface IndexedEdit { - edit: HashlineEdit; + edit: Edit; idx: number; } -type HashlineDeleteEdit = Extract; +type DeleteEdit = Extract; -interface HashlineReplacementGroup { +interface ReplacementGroup { startIndex: number; endIndex: number; sourceLineNum: number; replacement: string[]; - deletes: HashlineDeleteEdit[]; + deletes: DeleteEdit[]; } -function getHashlineEditAnchors(edit: HashlineEdit): Anchor[] { +function getEditAnchors(edit: Edit): Anchor[] { if (edit.kind === "delete") return [edit.anchor]; if (edit.cursor.kind === "before_anchor") return [edit.cursor.anchor]; if (edit.cursor.kind === "after_anchor") return [edit.cursor.anchor]; @@ -43,9 +46,9 @@ function getHashlineEditAnchors(edit: HashlineEdit): Anchor[] { * Verify every anchored edit points at an existing line. File-version binding is * checked once per section via the header hash before this function runs. */ -function validateHashlineLineBounds(edits: HashlineEdit[], fileLines: string[]): void { +function validateLineBounds(edits: Edit[], fileLines: string[]): void { for (const edit of edits) { - for (const anchor of getHashlineEditAnchors(edit)) { + for (const anchor of getEditAnchors(edit)) { if (anchor.line < 1 || anchor.line > fileLines.length) { throw new Error(`Line ${anchor.line} does not exist (file has ${fileLines.length} lines)`); } @@ -55,7 +58,7 @@ function validateHashlineLineBounds(edits: HashlineEdit[], fileLines: string[]): /** * Refuse a single-line replace whose target line is blank and whose payload is - * non-empty. The model is almost certainly miscounting: `A:CONTENT` overwrites + * non-empty. The author is almost certainly miscounting: `A:CONTENT` overwrites * the existing line, so applying it to a blank target deletes the blank cadence * and inserts content in its place. To insert content at a blank line, use * `A↑` (insert before) or `A↓` (insert after) instead. @@ -64,12 +67,12 @@ function validateHashlineLineBounds(edits: HashlineEdit[], fileLines: string[]): * `delete(A)` sharing the same source op line, no other inserts/deletes from * that op. */ -function detectReplaceOnBlankTarget(edits: HashlineEdit[], fileLines: string[]): string | null { - type Pair = { - insert?: Extract; - delete?: Extract; +function detectReplaceOnBlankTarget(edits: Edit[], fileLines: string[]): string | null { + interface Pair { + insert?: Extract; + delete?: Extract; multi?: boolean; - }; + } const byOpLine = new Map(); for (const edit of edits) { const pair = byOpLine.get(edit.lineNum) ?? {}; @@ -104,9 +107,9 @@ function detectReplaceOnBlankTarget(edits: HashlineEdit[], fileLines: string[]): return null; } -function insertAtStart(fileLines: string[], lineOrigins: HashlineLineOrigin[], lines: string[]): void { +function insertAtStart(fileLines: string[], lineOrigins: LineOrigin[], lines: string[]): void { if (lines.length === 0) return; - const origins = lines.map((): HashlineLineOrigin => "insert"); + const origins = lines.map((): LineOrigin => "insert"); if (fileLines.length === 1 && fileLines[0] === "") { fileLines.splice(0, 1, ...lines); lineOrigins.splice(0, 1, ...origins); @@ -116,9 +119,9 @@ function insertAtStart(fileLines: string[], lineOrigins: HashlineLineOrigin[], l lineOrigins.splice(0, 0, ...origins); } -function insertAtEnd(fileLines: string[], lineOrigins: HashlineLineOrigin[], lines: string[]): number | undefined { +function insertAtEnd(fileLines: string[], lineOrigins: LineOrigin[], lines: string[]): number | undefined { if (lines.length === 0) return undefined; - const origins = lines.map((): HashlineLineOrigin => "insert"); + const origins = lines.map((): LineOrigin => "insert"); if (fileLines.length === 1 && fileLines[0] === "") { fileLines.splice(0, 1, ...lines); lineOrigins.splice(0, 1, ...origins); @@ -132,14 +135,13 @@ function insertAtEnd(fileLines: string[], lineOrigins: HashlineLineOrigin[], lin } /** Bucket edits by the line they target so we can apply each line's group in one splice. */ - -function getAnchorTargetLine(edit: HashlineEdit): number | undefined { +function getAnchorTargetLine(edit: Edit): number | undefined { if (edit.kind === "delete") return edit.anchor.line; if (edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor") return edit.cursor.anchor.line; return undefined; } -function collectAnchorTargetLines(edits: HashlineEdit[]): Set { +function collectAnchorTargetLines(edits: Edit[]): Set { const lines = new Set(); for (const edit of edits) { const line = getAnchorTargetLine(edit); @@ -148,7 +150,7 @@ function collectAnchorTargetLines(edits: HashlineEdit[]): Set { return lines; } -function findReplacementGroup(edits: HashlineEdit[], startIndex: number): HashlineReplacementGroup | undefined { +function findReplacementGroup(edits: Edit[], startIndex: number): ReplacementGroup | undefined { const first = edits[startIndex]; if (first?.kind !== "insert" || first.cursor.kind !== "before_anchor") return undefined; @@ -162,7 +164,7 @@ function findReplacementGroup(edits: HashlineEdit[], startIndex: number): Hashli index++; } - const deletes: HashlineDeleteEdit[] = []; + const deletes: DeleteEdit[] = []; while (index < edits.length) { const edit = edits[index]; if (edit.kind !== "delete" || edit.lineNum !== sourceLineNum) break; @@ -231,10 +233,10 @@ interface DelimiterBalance { * Naive bracket counter — does NOT skip string/template/comment contents. The * single-line structural absorb relies on this being safe-by-asymmetry: the * candidate boundary line is constrained by `STRUCTURAL_CLOSING_BOUNDARY_RE` - * to be pure delimiters, so noise in deleted lines or non-boundary kept payload - * tends to push `expected !== kept` and biases the heuristic toward NOT - * absorbing (the safe direction). If we ever extend this to opening boundaries - * or non-structural single lines, swap this for a real tokenizer. + * to be pure delimiters, so noise in deleted lines or non-boundary kept + * payload tends to push `expected !== kept` and biases the heuristic toward + * NOT absorbing (the safe direction). If we ever extend this to opening + * boundaries or non-structural single lines, swap this for a real tokenizer. */ function computeDelimiterBalance(lines: string[]): DelimiterBalance { const balance: DelimiterBalance = { paren: 0, bracket: 0, brace: 0 }; @@ -364,7 +366,7 @@ function contiguousRange(start: number, count: number): number[] { return Array.from({ length: count }, (_, offset) => start + offset); } -function deleteEditForAutoAbsorbedLine(line: number, sourceLineNum: number, index: number): HashlineEdit { +function deleteEditForAutoAbsorbedLine(line: number, sourceLineNum: number, index: number): Edit { return { kind: "delete", anchor: { line }, @@ -373,15 +375,15 @@ function deleteEditForAutoAbsorbedLine(line: number, sourceLineNum: number, inde }; } -interface HashlinePureInsertGroup { +interface PureInsertGroup { startIndex: number; endIndex: number; sourceLineNum: number; - cursor: HashlineCursor; + cursor: Cursor; payload: string[]; } -function cursorMatches(a: HashlineCursor, b: HashlineCursor): boolean { +function cursorMatches(a: Cursor, b: Cursor): boolean { if (a.kind !== b.kind) return false; if (a.kind === "bof" || a.kind === "eof") return true; const aAnchor = (a as { anchor: Anchor }).anchor; @@ -396,7 +398,7 @@ function cursorMatches(a: HashlineCursor, b: HashlineCursor): boolean { * instead). Returns the contiguous payload so we can check it for boundary * duplicates against the file. */ -function findPureInsertGroup(edits: HashlineEdit[], startIndex: number): HashlinePureInsertGroup | undefined { +function findPureInsertGroup(edits: Edit[], startIndex: number): PureInsertGroup | undefined { const first = edits[startIndex]; if (first?.kind !== "insert") return undefined; @@ -429,10 +431,7 @@ function findPureInsertGroup(edits: HashlineEdit[], startIndex: number): Hashlin * - `belowStartIdx`: index of the first file line strictly below the * insertion point (`fileLines.length` if none). */ -function pureInsertNeighborhood( - cursor: HashlineCursor, - fileLines: string[], -): { aboveEndIdx: number; belowStartIdx: number } { +function pureInsertNeighborhood(cursor: Cursor, fileLines: string[]): { aboveEndIdx: number; belowStartIdx: number } { if (cursor.kind === "bof") return { aboveEndIdx: -1, belowStartIdx: 0 }; if (cursor.kind === "eof") return { aboveEndIdx: fileLines.length - 1, belowStartIdx: fileLines.length }; if (cursor.kind === "before_anchor") { @@ -458,7 +457,7 @@ interface PureInsertAbsorbResult { * duplicate absorption is enabled. */ function tryAbsorbPureInsertGroup( - group: HashlinePureInsertGroup, + group: PureInsertGroup, fileLines: string[], allowGenericBoundaryAbsorb: boolean, ): PureInsertAbsorbResult { @@ -521,13 +520,13 @@ function tryAbsorbPureInsertGroup( } function absorbReplacementBoundaryDuplicates( - edits: HashlineEdit[], + edits: Edit[], fileLines: string[], warnings: string[], - options: HashlineApplyOptions, -): HashlineEdit[] { + options: ApplyOptions, +): Edit[] { let nextSyntheticIndex = edits.length; - const absorbed: HashlineEdit[] = []; + const absorbed: Edit[] = []; // Anchor targets are stable across the loop because we only ever append // synthetic deletes (never mutate originals). A line in this set that @@ -672,15 +671,19 @@ function bucketAnchorEditsByLine(edits: IndexedEdit[]): Map "original"); + const lineOrigins: LineOrigin[] = fileLines.map(() => "original"); const warnings: string[] = []; let firstChangedLine: number | undefined; @@ -688,7 +691,7 @@ export function applyHashlineEdits( if (firstChangedLine === undefined || line < firstChangedLine) firstChangedLine = line; }; - validateHashlineLineBounds(edits, fileLines); + validateLineBounds(edits, fileLines); const blankTargetError = detectReplaceOnBlankTarget(edits, fileLines); if (blankTargetError !== null) throw new Error(blankTargetError); @@ -765,7 +768,7 @@ export function applyHashlineEdits( } } const replacement = deleteLine ? beforeLines : [...beforeLines, currentLine]; - const origins = replacement.map((): HashlineLineOrigin => (deleteLine ? "replacement" : "insert")); + const origins = replacement.map((): LineOrigin => (deleteLine ? "replacement" : "insert")); if (!deleteLine) { origins[origins.length - 1] = lineOrigins[idx] ?? "original"; } @@ -783,7 +786,7 @@ export function applyHashlineEdits( if (eofChangedLine !== undefined) trackFirstChanged(eofChangedLine); return { - lines: fileLines.join("\n"), + text: fileLines.join("\n"), firstChangedLine, ...(warnings.length > 0 ? { warnings } : {}), }; diff --git a/packages/coding-agent/src/hashline/diff-preview.ts b/packages/hashline/src/diff-preview.ts similarity index 51% rename from packages/coding-agent/src/hashline/diff-preview.ts rename to packages/hashline/src/diff-preview.ts index 624ebd39f..fdccb238c 100644 --- a/packages/coding-agent/src/hashline/diff-preview.ts +++ b/packages/hashline/src/diff-preview.ts @@ -1,18 +1,25 @@ -import type { CompactHashlineDiffOptions, CompactHashlineDiffPreview } from "./types"; +/** + * Re-number a unified diff that uses the `+|content` / + * `-|content` / ` |content` line format into a compact + * preview that anchors every line to its post-edit position. Added lines, + * removed lines, and context lines all end up with a hashline-style anchor + * so a follow-up edit can reuse them directly. + * + * This is intentionally decoupled from the diff producer: anything that + * emits the `|` shape works. + */ +import type { CompactDiffOptions, CompactDiffPreview } from "./types"; -export function buildCompactHashlineDiffPreview( - diff: string, - _options: CompactHashlineDiffOptions = {}, -): CompactHashlineDiffPreview { +export function buildCompactDiffPreview(diff: string, _options: CompactDiffOptions = {}): CompactDiffPreview { const lines = diff.length === 0 ? [] : diff.split("\n"); let addedLines = 0; let removedLines = 0; - // `generateDiffString` numbers `+` lines with the post-edit line number, + // External diff producers number `+` lines with the post-edit line number, // `-` lines with the pre-edit line number, and context lines with the - // pre-edit line number. To emit fresh line numbers usable for follow-up edits, - // we convert context-line numbers to post-edit positions by tracking the - // running offset (added so far - removed so far) as we walk the diff. + // pre-edit line number. To emit fresh line numbers usable for follow-up + // edits, convert context-line numbers to post-edit positions by tracking + // the running offset (added so far - removed so far) as we walk the diff. const formatted = lines.map(line => { const kind = line[0]; if (kind !== "+" && kind !== "-" && kind !== " ") return line; diff --git a/packages/coding-agent/src/hashline/hash.ts b/packages/hashline/src/format.ts similarity index 67% rename from packages/coding-agent/src/hashline/hash.ts rename to packages/hashline/src/format.ts index fa143c228..ba0925c12 100644 --- a/packages/coding-agent/src/hashline/hash.ts +++ b/packages/hashline/src/format.ts @@ -1,16 +1,43 @@ /** - * Core hash utilities shared by hashline edit mode, read/search output, - * and prompt helpers. + * Hashline format primitives: sigils, separators, regex fragments, and the + * file-hash computation. These are the single source of truth for the + * parser, the tokenizer, the prompt, and the formal grammar. */ -const regexEscape = (str: string): string => str.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +/** Op sigil used immediately after a line-number anchor to insert before it. */ +export const HL_OP_INSERT_BEFORE = "↑"; +/** Op sigil used immediately after a line-number anchor to insert after it. */ +export const HL_OP_INSERT_AFTER = "↓"; +/** Op sigil used after a range (or single anchor) to replace its lines. */ +export const HL_OP_REPLACE = ":"; +/** Op sigil used after a range (or single anchor) to delete its lines. */ +export const HL_OP_DELETE = "!"; + +/** All hashline edit op sigils, concatenated for fast membership tests. */ +export const HL_OP_CHARS = `${HL_OP_INSERT_BEFORE}${HL_OP_INSERT_AFTER}${HL_OP_REPLACE}${HL_OP_DELETE}`; + +/** Prefix for payload continuation lines. The prefix itself is not written. */ +export const HL_PAYLOAD_PREFIX = "+"; + +/** Hashline edit file-section header marker. */ +export const HL_FILE_PREFIX = "¶"; + +/** Separator between a hashline file path and its file hash. */ +export const HL_FILE_HASH_SEP = "#"; + +/** Separator between a line number and displayed line content in hashline mode. */ +export const HL_LINE_BODY_SEP = ":"; + +function regexEscape(str: string): string { + return str.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +} /** * Decoration prefix that may precede a line number in tool output: * `>` (context line in grep), `-` (removed line), `*` (match line). - * Any combination, in any order, surrounded by optional - * whitespace. Output formatters emit at most one decoration per line; the - * parser stays liberal because it accepts whatever the model echoes back. + * Any combination, in any order, surrounded by optional whitespace. Output + * formatters emit at most one decoration per line; the parser stays liberal + * because it accepts whatever the model echoes back. */ export const HL_ANCHOR_DECORATION_RE_RAW = `\\s*[>\\-*]*\\s*`; @@ -29,12 +56,6 @@ export const HL_FILE_HASH_RE_RAW = `[0-9a-f]{4}`; /** Capture-group form of {@link HL_FILE_HASH_RE_RAW}. */ export const HL_FILE_HASH_CAPTURE_RE_RAW = `(${HL_FILE_HASH_RE_RAW})`; -/** Separator between a hashline file path and its file hash. */ -export const HL_FILE_HASH_SEP = "#"; - -/** Separator between a line number and displayed line content in hashline mode. */ -export const HL_LINE_BODY_SEP = ":"; - /** Regex-escaped form of {@link HL_LINE_BODY_SEP}, safe for embedding inside a regex. */ export const HL_LINE_BODY_SEP_RE_RAW = regexEscape(HL_LINE_BODY_SEP); @@ -53,48 +74,6 @@ export function describeAnchorExamples(linePrefix = ""): string { return examples.map(e => `"${e}"`).join(", "); } -/** - * Substitute every grammar placeholder with the value derived from its - * TypeScript counterpart. Grammars that don't reference these placeholders - * pass through unchanged. - */ -export function resolveHashlineGrammarPlaceholders(grammar: string): string { - return grammar - .replaceAll("$HFMT$", "") - .replaceAll("$HFILE_HASH$", HL_FILE_HASH_RE_RAW) - .replaceAll("$HFILE_HASH_SEP$", HL_FILE_HASH_SEP) - .replaceAll("$HOP_INSERT_BEFORE$", HL_OP_INSERT_BEFORE) - .replaceAll("$HOP_INSERT_AFTER$", HL_OP_INSERT_AFTER) - .replaceAll("$HOP_REPLACE$", HL_OP_REPLACE) - .replaceAll("$HOP_DELETE$", HL_OP_DELETE) - .replaceAll("$HOP_CHARS$", HL_OP_CHARS) - .replaceAll("$HFILE$", HL_FILE_PREFIX); -} - -/** - * op lines have an `ANCHOR[INLINE_PAYLOAD]` shape, where SIGIL is one of - * {@link HL_OP_INSERT_BEFORE}, {@link HL_OP_INSERT_AFTER}, {@link HL_OP_REPLACE}, - * or {@link HL_OP_DELETE}. Multi-line payloads follow on subsequent lines - * prefixed with {@link HL_PAYLOAD_PREFIX}; that prefix is stripped before the - * payload is written. - * - * These constants are the single source of truth for the edit parser, grammar, - * renderer, and prompt. - */ -export const HL_OP_INSERT_BEFORE = "↑"; -export const HL_OP_INSERT_AFTER = "↓"; -export const HL_OP_REPLACE = ":"; -export const HL_OP_DELETE = "!"; - -/** Prefix for payload continuation lines. The prefix itself is not written. */ -export const HL_PAYLOAD_PREFIX = "+"; - -/** All hashline edit op sigils, concatenated for fast membership tests. */ -export const HL_OP_CHARS = `${HL_OP_INSERT_BEFORE}${HL_OP_INSERT_AFTER}${HL_OP_REPLACE}${HL_OP_DELETE}`; - -/** Hashline edit file section header marker. */ -export const HL_FILE_PREFIX = "¶"; - function normalizeFileHashText(text: string): string { return text .replace(/\r/g, "") @@ -104,8 +83,8 @@ function normalizeFileHashText(text: string): string { } /** - * Compute the 4-hex-character hash carried by a hashline section header. - * The hash normalizes CR characters and trailing whitespace before hashing so + * Compute the 4-hex-character hash carried by a hashline section header. The + * hash normalizes CR characters and trailing whitespace before hashing so * platform line endings and display-trimmed lines do not invalidate anchors. */ export function computeFileHash(text: string): string { diff --git a/packages/hashline/src/fs.ts b/packages/hashline/src/fs.ts new file mode 100644 index 000000000..b11ed904c --- /dev/null +++ b/packages/hashline/src/fs.ts @@ -0,0 +1,159 @@ +/** + * Storage seam for the hashline patcher. {@link Filesystem} is intentionally + * minimal — `readText`, `writeText`, `exists` — so any backing store can be + * adapted: disk, memory, S3, an LSP text-document protocol, a Git tree, a + * VFS, etc. + * + * The patcher does its own BOM stripping and LF normalization between + * {@link Filesystem.readText} and {@link Filesystem.writeText}; the FS deals + * only in raw text strings. + */ + +/** + * Result returned by {@link Filesystem.writeText}. The patcher echoes back + * `text` so adapters that transform on serialization (e.g. notebooks) can + * report what actually landed on disk. + */ +export interface WriteResult { + /** Final text that was persisted. May differ from the input if the FS transformed it. */ + text: string; +} + +/** + * ENOENT-like error thrown by {@link Filesystem.readText} when a path is + * missing. Carrying a `code` property keeps the contract compatible with + * `node:fs` callers that already check `err.code === "ENOENT"`. + */ +export class NotFoundError extends Error { + readonly code = "ENOENT"; + + constructor(path: string, cause?: unknown) { + super(`File not found: ${path}`); + this.name = "NotFoundError"; + if (cause !== undefined) (this as Error & { cause?: unknown }).cause = cause; + } +} + +/** Type guard for {@link NotFoundError} and structurally-compatible errors. */ +export function isNotFound(error: unknown): boolean { + if (error instanceof NotFoundError) return true; + if (error instanceof Error && (error as Error & { code?: string }).code === "ENOENT") return true; + return false; +} + +/** + * Abstract storage backend the {@link Patcher} reads from and writes to. + * Subclass for new backends; the package ships {@link InMemoryFilesystem} and + * {@link NodeFilesystem} for the most common cases. + * + * Implementations work with raw text — the patcher handles BOM stripping and + * line-ending normalization itself. `readText` MUST throw {@link + * NotFoundError} (or any error for which {@link isNotFound} returns true) + * when the path doesn't exist; that's how the patcher detects a create-vs- + * update. + */ +export abstract class Filesystem { + /** Read the file's full text content. Throw on missing file. */ + abstract readText(path: string): Promise; + + /** Persist `content` at `path`. Returns the actual final text that was written. */ + abstract writeText(path: string, content: string): Promise; + + /** Return true when the path exists and can be read. Default: probe via {@link readText}. */ + async exists(path: string): Promise { + try { + await this.readText(path); + return true; + } catch (error) { + if (isNotFound(error)) return false; + throw error; + } + } + + /** + * Canonical path used as a key by external caches (e.g. snapshot + * stores). The default is identity; override to return an absolute or + * otherwise canonicalised path so producers and consumers of cached + * snapshots agree on the key without each having to redo the resolution. + */ + canonicalPath(path: string): string { + return path; + } +} + +/** + * In-memory {@link Filesystem}. Useful for tests, sandboxes, dry-runs, and as + * a building block for stacked adapters (e.g. an LRU layer on top). + */ +export class InMemoryFilesystem extends Filesystem { + #files = new Map(); + + constructor(initial?: Iterable) { + super(); + if (initial) { + for (const [path, content] of initial) this.#files.set(path, content); + } + } + + async readText(path: string): Promise { + const text = this.#files.get(path); + if (text === undefined) throw new NotFoundError(path); + return text; + } + + async writeText(path: string, content: string): Promise { + this.#files.set(path, content); + return { text: content }; + } + + async exists(path: string): Promise { + return this.#files.has(path); + } + + /** Synchronous helper for setting up fixtures without awaiting. */ + set(path: string, content: string): void { + this.#files.set(path, content); + } + + /** Synchronous helper for inspecting state without awaiting. */ + get(path: string): string | undefined { + return this.#files.get(path); + } + + /** Remove a single entry. Returns true when something was removed. */ + delete(path: string): boolean { + return this.#files.delete(path); + } + + /** Wipe all entries. */ + clear(): void { + this.#files.clear(); + } + + /** Iterate `[path, content]` pairs. */ + entries(): IterableIterator<[string, string]> { + return this.#files.entries(); + } +} + +/** + * Disk-backed {@link Filesystem} using Bun's file APIs. The default for CLI + * use. Paths are accepted as-is; callers responsible for any cwd or + * jail/sandbox resolution should wrap this with their own subclass. + */ +export class NodeFilesystem extends Filesystem { + async readText(path: string): Promise { + const file = Bun.file(path); + if (!(await file.exists())) throw new NotFoundError(path); + return file.text(); + } + + async writeText(path: string, content: string): Promise { + await Bun.write(path, content); + return { text: content }; + } + + async exists(path: string): Promise { + return Bun.file(path).exists(); + } +} diff --git a/packages/coding-agent/src/hashline/grammar.lark b/packages/hashline/src/grammar.lark similarity index 57% rename from packages/coding-agent/src/hashline/grammar.lark rename to packages/hashline/src/grammar.lark index 13d9f92bf..9f0ef23ad 100644 --- a/packages/coding-agent/src/hashline/grammar.lark +++ b/packages/hashline/src/grammar.lark @@ -3,16 +3,16 @@ begin_patch: "*** Begin Patch" LF end_patch: "*** End Patch" LF? hunk: update_hunk -update_hunk: "$HFILE$" filename ("#" file_hash)? LF line_op* +update_hunk: "¶" filename ("#" file_hash)? LF line_op* filename: /([^\s#]+)/ file_hash: /[0-9a-f]{4}/ line_op: insert_before | insert_after | replace | delete -insert_before: anchor "$HOP_INSERT_BEFORE$" LF payload* -insert_after: anchor "$HOP_INSERT_AFTER$" LF payload* -replace: range "$HOP_REPLACE$" LF payload* -delete: range "$HOP_DELETE$" LF +insert_before: anchor "↑" LF payload* +insert_after: anchor "↓" LF payload* +replace: range ":" LF payload* +delete: range "!" LF payload: "+" /[^\n]*/ LF anchor: LID | "EOF" | "BOF" diff --git a/packages/coding-agent/src/hashline/index.ts b/packages/hashline/src/index.ts similarity index 50% rename from packages/coding-agent/src/hashline/index.ts rename to packages/hashline/src/index.ts index 1e3c264a0..96e7696c2 100644 --- a/packages/coding-agent/src/hashline/index.ts +++ b/packages/hashline/src/index.ts @@ -1,14 +1,16 @@ -export * from "./anchors"; export * from "./apply"; -export * from "./constants"; -export * from "./diff"; export * from "./diff-preview"; -export * from "./execute"; -export * from "./executor"; -export * from "./hash"; +export * from "./format"; +export * from "./fs"; export * from "./input"; +export * from "./messages"; +export * from "./mismatch"; +export * from "./normalize"; +export * from "./parser"; +export * from "./patcher"; export * from "./prefixes"; export * from "./recovery"; +export * from "./snapshots"; export * from "./stream"; export * from "./tokenizer"; export * from "./types"; diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts new file mode 100644 index 000000000..78edb9faa --- /dev/null +++ b/packages/hashline/src/input.ts @@ -0,0 +1,319 @@ +/** + * Top-level patch parser. Splits an authored hashline input into a list of + * {@link PatchSection}s, each rooted at a `¶PATH#HASH` header, then exposes + * a {@link Patch} class that gives lazy access to the parsed edits per + * section. + * + * The splitter is purely lexical — it doesn't know whether a section's path + * actually exists. That's the patcher's job. + */ +import * as path from "node:path"; +import { applyEdits } from "./apply"; +import { HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import { parsePatch } from "./parser"; +import { Tokenizer } from "./tokenizer"; +import type { ApplyOptions, ApplyResult, Edit, SplitOptions } from "./types"; + +// Pure classification — single shared tokenizer is safe. +const TOKENIZER = new Tokenizer(); + +function unquoteHashlinePath(pathText: string): string { + if (pathText.length < 2) return pathText; + const first = pathText[0]; + const last = pathText[pathText.length - 1]; + if ((first === '"' || first === "'") && first === last) return pathText.slice(1, -1); + return pathText; +} + +function normalizeHashlinePath(rawPath: string, cwd?: string): string { + const unquoted = unquoteHashlinePath(rawPath.trim()); + if (!cwd || !path.isAbsolute(unquoted)) return unquoted; + const relative = path.relative(path.resolve(cwd), path.resolve(unquoted)); + const isWithinCwd = relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative)); + return isWithinCwd ? relative || "." : unquoted; +} + +interface RawSection { + path: string; + fileHash?: string; + diff: string; +} + +/** + * Parse a `¶PATH[#hash]` header line. Returns `null` for lines that do not + * begin with the `¶` prefix; throws the existing "Input header must be …" + * error when a `¶`-prefixed line fails the strict shape (so malformed paths + * surface immediately instead of being silently re-classified as payload). + */ +function parseHashlineHeaderLine(line: string, cwd?: string): RawSection | null { + const trimmed = line.trimEnd(); + if (!trimmed.startsWith(HL_FILE_PREFIX)) return null; + + const token = TOKENIZER.tokenize(trimmed); + if (token.kind !== "header") { + throw new Error( + `Input header must be ${HL_FILE_PREFIX}PATH or ${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}HASH with a 4-hex file hash; got ${JSON.stringify(trimmed)}.`, + ); + } + + const parsedPath = normalizeHashlinePath(token.path, cwd); + if (parsedPath.length === 0) { + throw new Error(`Input header "${HL_FILE_PREFIX}" is empty; provide a file path.`); + } + return token.fileHash !== undefined + ? { path: parsedPath, fileHash: token.fileHash, diff: "" } + : { path: parsedPath, diff: "" }; +} + +function stripLeadingBlankLines(input: string): string { + const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input; + const lines = stripped.split("\n"); + while (lines.length > 0) { + const head = lines[0].replace(/\r$/, ""); + if (head.trim().length === 0 || TOKENIZER.tokenize(head).kind === "envelope-begin") { + lines.shift(); + continue; + } + break; + } + return lines.join("\n"); +} + +/** + * Returns true when the input contains at least one line that the tokenizer + * recognizes as a hashline op. Used by streaming previews to decide whether + * the partial input is worth treating as a hashline patch yet. + */ +export function containsRecognizableHashlineOperations(input: string): boolean { + for (const line of input.split(/\r?\n/)) { + if (TOKENIZER.isOp(line)) return true; + } + return false; +} + +function normalizeFallbackInput(input: string, options: SplitOptions): string { + const stripped = input.startsWith("\uFEFF") ? input.slice(1) : input; + const hasExplicitHeader = stripped + .split(/\r?\n/) + .some(rawLine => parseHashlineHeaderLine(rawLine, options.cwd) !== null); + if (hasExplicitHeader) return input; + + if (!options.path || !containsRecognizableHashlineOperations(input)) return input; + const fallbackPath = normalizeHashlinePath(options.path, options.cwd); + if (fallbackPath.length === 0) return input; + return `${HL_FILE_PREFIX}${fallbackPath}\n${input}`; +} + +function splitRawSections(input: string, options: SplitOptions = {}): RawSection[] { + const stripped = stripLeadingBlankLines(normalizeFallbackInput(input, options)); + const lines = stripped.split(/\r?\n/); + const firstLine = lines[0] ?? ""; + + if (parseHashlineHeaderLine(firstLine, options.cwd) === null) { + const preview = JSON.stringify(firstLine.slice(0, 120)); + throw new Error( + `input must begin with "${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}HASH" on the first non-blank line for anchored edits; got: ${preview}. ` + + `Example: "${HL_FILE_PREFIX}src/foo.ts${HL_FILE_HASH_SEP}1a2b" then edit ops.`, + ); + } + + const sections: RawSection[] = []; + let current: RawSection | undefined; + let currentLines: string[] = []; + + const flush = () => { + if (!current) return; + const hasOps = currentLines.some(line => line.trim().length > 0); + if (hasOps) sections.push({ ...current, diff: currentLines.join("\n") }); + currentLines = []; + }; + + for (const line of lines) { + const trimmed = line.trimEnd(); + const token = TOKENIZER.tokenize(line); + if (token.kind === "envelope-end" || token.kind === "abort") break; + if (token.kind === "envelope-begin") continue; + + // Route every `¶`-prefixed line through parseHashlineHeaderLine so + // malformed headers still raise the strict "Input header must be …" + // diagnostic (the tokenizer alone would silently classify them as + // payload). + if (trimmed.startsWith(HL_FILE_PREFIX)) { + const header = parseHashlineHeaderLine(line, options.cwd); + if (header !== null) { + flush(); + current = header; + currentLines = []; + continue; + } + } + currentLines.push(line); + } + flush(); + return sections; +} + +/** + * Snapshot of one section in a parsed {@link Patch}: a target file plus the + * lazily-parsed list of edits that should land on it. Constructed by + * {@link Patch.parse}; consumers usually iterate `patch.sections` rather + * than build these directly. + */ +export class PatchSection { + readonly path: string; + readonly fileHash: string | undefined; + readonly diff: string; + #parsed: { edits: Edit[]; warnings: string[] } | undefined; + + constructor(raw: RawSection) { + this.path = raw.path; + this.fileHash = raw.fileHash; + this.diff = raw.diff; + } + + /** + * Parse this section's diff body. Cached: subsequent calls return the + * same `{ edits, warnings }` object so callers can safely call this from + * multiple paths (preflight, apply, diff-preview). + */ + parse(): { edits: Edit[]; warnings: readonly string[] } { + this.#parsed ??= parsePatch(this.diff); + return this.#parsed; + } + + /** Parsed edits for this section. */ + get edits(): readonly Edit[] { + return this.parse().edits; + } + + /** Warnings emitted during parsing of this section. */ + get warnings(): readonly string[] { + return this.parse().warnings; + } + + /** + * True when at least one edit anchors to a concrete file line (range or + * before/after_anchor insert). Pure BOF/EOF inserts do not count: those + * are safe to apply to files that don't yet exist. + */ + get hasAnchorScopedEdit(): boolean { + return this.edits.some(edit => { + if (edit.kind === "delete") return true; + return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor"; + }); + } + + /** Anchor lines touched by this section, sorted ascending and deduplicated. */ + collectAnchorLines(): readonly number[] { + const lines = new Set(); + for (const edit of this.edits) { + if (edit.kind === "delete") { + lines.add(edit.anchor.line); + continue; + } + if (edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor") { + lines.add(edit.cursor.anchor.line); + } + } + return [...lines].sort((a, b) => a - b); + } + + /** + * Apply this section's edits to `text` and return the post-edit result. + * Pure: does no I/O, does not validate the section file hash. The + * {@link Patcher} owns hash validation and recovery; reach for this + * method directly when you've already validated the file content and + * just want the result. + */ + applyTo(text: string, options: ApplyOptions = {}): ApplyResult { + const { edits, warnings } = this.parse(); + const result = applyEdits(text, [...edits], options); + // Preserve parse warnings alongside applier warnings so consumers + // don't need to call `parse()` separately. + const merged = warnings.length === 0 ? result.warnings : [...warnings, ...(result.warnings ?? [])]; + return merged && merged.length > 0 + ? { ...result, warnings: merged } + : { text: result.text, firstChangedLine: result.firstChangedLine }; + } +} + +/** + * A parsed hashline patch — zero or more {@link PatchSection}s, each rooted + * at a `¶PATH#HASH` header. Construct via {@link Patch.parse}. + * + * `Patch` is pure data: parsing is line-anchored and does not look at the + * filesystem. To apply a patch, hand it to {@link Patcher.apply}. + */ +export class Patch { + readonly sections: readonly PatchSection[]; + + private constructor(sections: PatchSection[]) { + this.sections = sections; + } + + /** + * Parse `input` into a {@link Patch}. `options.cwd` resolves absolute + * paths inside headers to cwd-relative form; `options.path` provides a + * fallback when the input lacks a header but contains hashline ops + * (useful for streaming previews). + * + * Consecutive sections targeting the same path are merged into a single + * section with concatenated diff bodies. Anchors authored against the + * same file snapshot must be applied as one batch; otherwise the first + * sub-edit shifts line numbers out from under the second's anchors and + * validation fails. + */ + static parse(input: string, options: SplitOptions = {}): Patch { + const raw = mergeSamePathSections(splitRawSections(input, options)); + return new Patch(raw.map(section => new PatchSection(section))); + } + + /** + * Parse `input` and return only the first section. Throws if the input + * has zero sections. Convenience for the single-section case where the + * caller already knows the patch is one hunk. + */ + static parseSingle(input: string, options: SplitOptions = {}): PatchSection { + const patch = Patch.parse(input, options); + const first = patch.sections[0]; + if (!first) throw new Error("Patch input did not produce any sections."); + return first; + } +} + +/** + * Collapse consecutive or interleaved sections targeting the same path into a + * single section with concatenated diffs. Anchors authored against the same + * file snapshot must be applied as one batch; otherwise the first sub-edit + * shifts line numbers out from under the second's anchors and validation + * fails. Path order is preserved by first occurrence. + */ +function mergeSamePathSections(sections: RawSection[]): RawSection[] { + const byPath = new Map(); + for (const section of sections) { + const existing = byPath.get(section.path); + if (existing) { + if ( + existing.fileHash !== undefined && + section.fileHash !== undefined && + existing.fileHash !== section.fileHash + ) { + throw new Error( + `Conflicting hashline file hashes for ${section.path}: #${existing.fileHash} and #${section.fileHash}. Re-read the file and retry with one current header.`, + ); + } + if (existing.fileHash === undefined && section.fileHash !== undefined) existing.fileHash = section.fileHash; + existing.diffs.push(section.diff); + continue; + } + byPath.set(section.path, { + ...(section.fileHash !== undefined ? { fileHash: section.fileHash } : {}), + diffs: [section.diff], + }); + } + return Array.from(byPath, ([sectionPath, entry]) => ({ + path: sectionPath, + ...(entry.fileHash !== undefined ? { fileHash: entry.fileHash } : {}), + diff: entry.diffs.join("\n"), + })); +} diff --git a/packages/coding-agent/src/hashline/constants.ts b/packages/hashline/src/messages.ts similarity index 59% rename from packages/coding-agent/src/hashline/constants.ts rename to packages/hashline/src/messages.ts index 5854b85df..870286aeb 100644 --- a/packages/coding-agent/src/hashline/constants.ts +++ b/packages/hashline/src/messages.ts @@ -1,3 +1,10 @@ +/** + * Centralized error and warning text emitted by the hashline parser, applier, + * and patcher. Consolidating these as named constants makes them easy to + * audit and keeps wording stable across the rendering paths that surface + * them. + */ + /** Lines of context shown either side of a hash mismatch. */ export const MISMATCH_CONTEXT = 2; @@ -8,30 +15,28 @@ export const BEGIN_PATCH_MARKER = "*** Begin Patch"; export const END_PATCH_MARKER = "*** End Patch"; /** - * Recovery sentinel emitted by the agent loop when a contaminated - * `to=functions.edit` stream is truncated mid-call (see - * `docs/ERRATA-GPT5-HARMONY.md`). Behaves like `END_PATCH_MARKER` for - * parsing — terminates the line loop — and additionally surfaces a - * warning in the tool result so the model knows to re-issue any - * remaining edits. + * Recovery sentinel emitted by an agent loop when a contaminated tool-call + * stream is truncated mid-call. Behaves like {@link END_PATCH_MARKER} for + * parsing — terminates the line loop — and additionally surfaces a warning + * so the caller knows to re-issue any remaining edits. */ export const ABORT_MARKER = "*** Abort"; -/** Warning text appended to the tool result when ABORT_MARKER terminates parsing. */ +/** Warning text appended to the tool result when {@link ABORT_MARKER} terminates parsing. */ export const ABORT_WARNING = "Tool stream truncated mid-call due to detected output corruption. Applied ops above are valid. Re-issue any remaining edits."; /** * Warning text appended when two consecutive `A-B:` ops on the exact same - * range get coalesced (model painted a before/after pair). The second op - * wins; the first op's payload is silently discarded. + * range get coalesced (model painted a before/after pair). The second op wins; + * the first op's payload is silently discarded. */ export const REPLACE_PAIR_COALESCED_WARNING = "Detected an identical-range before/after replace pair; kept only the second block's payload. Issue ONE op per range — the payload is the final desired content, never both old and new."; /** * Warning text appended when un-prefixed continuation lines are accepted as - * implicit payload (lenient legacy behavior). The model authored a multi-line + * implicit payload (lenient legacy behavior). The author wrote a multi-line * replace without `+` prefixes; the parser accepted it because the lines did * not classify as ops/headers/payloads, but the canonical syntax requires `+` * on every continuation line after the op. @@ -42,19 +47,30 @@ export const IMPLICIT_CONTINUATION_WARNING = /** * Warning text appended when an inner `LINE:TEXT` (or sub-range `A-B:TEXT`) * op arrives while an outer `A-B:` replace is still pending and the inner - * anchor falls inside the outer range. The model used the read-output + * anchor falls inside the outer range. The author used the read-output * `LINE:TEXT` format as if it were a payload-continuation line; we strip the * `LINE:` prefix and append the body to the pending payload, but warn so the * canonical `+`-continuation form remains preferred. */ export const PAYLOAD_LINE_PREFIX_DEMOTED_WARNING = "Detected one or more `LINE:TEXT` lines whose anchors fell inside a pending replace range; treated them as payload-continuation lines and stripped the `LINE:` prefix. Inside an `A-B:` block, every payload line must be on its own row prefixed with `+` — never reuse the read-output gutter format."; + /** * Warning text appended when an op carries an inline payload (`LINE:TEXT`, - * `A-B:TEXT`, `LINE↑TEXT`, `LINE↓TEXT`). Canonical syntax is bare op + - * `+`-prefixed continuation rows; we accept the inline form leniently so the - * model's first-attempt edit still lands, but warn so the canonical form - * remains preferred. + * `LINE↑CONTENT`, `LINE↓CONTENT`). Canonical syntax is the bare op followed + * by `+`-prefixed payload rows on the next line(s). */ export const INLINE_PAYLOAD_ACCEPTED_WARNING = "Accepted inline payload on the op line (e.g. `LINE:CONTENT`, `LINE↑CONTENT`). Canonical syntax is the bare op followed by `+`-prefixed payload rows on the next line(s). Prefer the explicit form."; + +/** Warning text emitted by `Recovery` when an external write fits a cached snapshot. */ +export const RECOVERY_EXTERNAL_WARNING = + "Recovered from a stale file hash using a previous read snapshot (file changed externally between read and edit)."; + +/** Warning text emitted by `Recovery` when a prior in-session edit advanced the hash. */ +export const RECOVERY_SESSION_CHAIN_WARNING = + "Recovered from a stale file hash using an earlier in-session snapshot (the file hash advanced after a prior edit in this session)."; + +/** Warning text emitted by `Recovery` when the session-chain fast-path was taken. */ +export const RECOVERY_SESSION_REPLAY_WARNING = + "Recovered by replaying your edits onto the current file content — your previous edit in this session changed line(s) you re-targeted with a stale hash. Verify the diff matches your intent before continuing."; diff --git a/packages/coding-agent/src/hashline/anchors.ts b/packages/hashline/src/mismatch.ts similarity index 69% rename from packages/coding-agent/src/hashline/anchors.ts rename to packages/hashline/src/mismatch.ts index b3c2ed2ca..274b22044 100644 --- a/packages/coding-agent/src/hashline/anchors.ts +++ b/packages/hashline/src/mismatch.ts @@ -1,8 +1,17 @@ -import { MISMATCH_CONTEXT } from "./constants"; -import { formatNumberedLine, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./hash"; +/** + * Error type raised when a section's file-hash does not match the live file + * content and recovery is unavailable / has failed. + * + * Carries enough context to render a useful diagnostic: the anchored lines + * plus a couple of lines of surrounding context. The {@link MismatchError} + * formats this into a message at construction time. + */ +import { formatNumberedLine, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import { MISMATCH_CONTEXT } from "./messages"; const LINE_REF_RE = /^\s*[>+\-*]*\s*(\d+)(?::.*)?\s*$/; +/** Format the required-shape diagnostic shown when a line reference is malformed. */ export function formatFullAnchorRequirement(raw?: string): string { const received = raw === undefined ? "" : ` Received ${JSON.stringify(raw)}.`; return ( @@ -11,6 +20,7 @@ export function formatFullAnchorRequirement(raw?: string): string { ); } +/** Parse a decorated bare line-number anchor like `42`, `*42:foo`, ` > 7`. */ export function parseTag(ref: string): { line: number } { const match = ref.match(LINE_REF_RE); if (!match) { @@ -21,7 +31,7 @@ export function parseTag(ref: string): { line: number } { return { line }; } -export interface HashlineMismatchDetails { +export interface MismatchDetails { path?: string; expectedFileHash: string; actualFileHash: string; @@ -40,16 +50,22 @@ function getMismatchDisplayLines(anchorLines: readonly number[], fileLines: stri return [...displayLines].sort((a, b) => a - b); } -export class HashlineMismatchError extends Error { +/** + * Raised when a hashline section's file hash doesn't match the live file's + * content (and recovery, if configured, declined the merge). Carries the + * file lines plus anchored lines so renderers can produce a richer + * diagnostic via {@link MismatchError.displayMessage}. + */ +export class MismatchError extends Error { readonly path: string | undefined; readonly expectedFileHash: string; readonly actualFileHash: string; readonly fileLines: string[]; readonly anchorLines: readonly number[]; - constructor(details: HashlineMismatchDetails) { - super(HashlineMismatchError.formatMessage(details)); - this.name = "HashlineMismatchError"; + constructor(details: MismatchDetails) { + super(MismatchError.formatMessage(details)); + this.name = "MismatchError"; this.path = details.path; this.expectedFileHash = details.expectedFileHash; this.actualFileHash = details.actualFileHash; @@ -58,7 +74,7 @@ export class HashlineMismatchError extends Error { } get displayMessage(): string { - return HashlineMismatchError.formatDisplayMessage({ + return MismatchError.formatDisplayMessage({ path: this.path, expectedFileHash: this.expectedFileHash, actualFileHash: this.actualFileHash, @@ -67,7 +83,7 @@ export class HashlineMismatchError extends Error { }); } - static rejectionHeader(details: HashlineMismatchDetails): string[] { + static rejectionHeader(details: MismatchDetails): string[] { const pathText = details.path ? ` for ${details.path}` : ""; return [ `Edit rejected${pathText}: file changed between read and edit.`, @@ -75,13 +91,13 @@ export class HashlineMismatchError extends Error { ]; } - static formatDisplayMessage(details: HashlineMismatchDetails): string { - return HashlineMismatchError.formatMessage(details); + static formatDisplayMessage(details: MismatchDetails): string { + return MismatchError.formatMessage(details); } - static formatMessage(details: HashlineMismatchDetails): string { + static formatMessage(details: MismatchDetails): string { const anchorSet = new Set(details.anchorLines ?? []); - const lines = HashlineMismatchError.rejectionHeader(details); + const lines = MismatchError.rejectionHeader(details); const displayLines = getMismatchDisplayLines(details.anchorLines ?? [], details.fileLines); if (displayLines.length === 0) return lines.join("\n"); lines.push(""); @@ -97,6 +113,7 @@ export class HashlineMismatchError extends Error { } } +/** Throws when the line reference is out of bounds for the given file. */ export function validateLineRef(ref: { line: number }, fileLines: string[]): void { if (ref.line < 1 || ref.line > fileLines.length) { throw new Error(`Line ${ref.line} does not exist (file has ${fileLines.length} lines)`); diff --git a/packages/hashline/src/normalize.ts b/packages/hashline/src/normalize.ts new file mode 100644 index 000000000..ab7cd40c6 --- /dev/null +++ b/packages/hashline/src/normalize.ts @@ -0,0 +1,39 @@ +/** + * Minimal text-shape normalization: line-ending detection / round-trip and + * BOM stripping. The patcher uses these to canonicalize text to LF before + * applying edits and to restore the original shape on write-back. + */ + +export type LineEnding = "\r\n" | "\n"; + +/** Detect the predominant line ending in `content`. Defaults to LF when neither is present. */ +export function detectLineEnding(content: string): LineEnding { + const crlfMatches = content.match(/\r\n/g); + const crlfCount = crlfMatches ? crlfMatches.length : 0; + const lfMatches = content.match(/\n/g); + const lfCount = lfMatches ? lfMatches.length : 0; + const standaloneLfCount = lfCount - crlfCount; + return crlfCount > standaloneLfCount ? "\r\n" : "\n"; +} + +/** Normalize every line ending to LF. */ +export function normalizeToLF(text: string): string { + return text.replace(/\r\n?/g, "\n"); +} + +/** Re-encode LF text with the requested line ending. */ +export function restoreLineEndings(text: string, ending: LineEnding): string { + return ending === "\r\n" ? text.replace(/\n/g, "\r\n") : text; +} + +export interface BomResult { + /** Either the empty string or the BOM sequence (currently UTF-8 BOM). */ + bom: string; + /** Text with any leading BOM removed. */ + text: string; +} + +/** Strip a UTF-8 BOM if present and return both the BOM and the trailing text. */ +export function stripBom(content: string): BomResult { + return content.startsWith("\uFEFF") ? { bom: "\uFEFF", text: content.slice(1) } : { bom: "", text: content }; +} diff --git a/packages/coding-agent/src/hashline/executor.ts b/packages/hashline/src/parser.ts similarity index 83% rename from packages/coding-agent/src/hashline/executor.ts rename to packages/hashline/src/parser.ts index c38019364..22951882d 100644 --- a/packages/coding-agent/src/hashline/executor.ts +++ b/packages/hashline/src/parser.ts @@ -1,10 +1,18 @@ -import { - ABORT_WARNING, - IMPLICIT_CONTINUATION_WARNING, - INLINE_PAYLOAD_ACCEPTED_WARNING, - PAYLOAD_LINE_PREFIX_DEMOTED_WARNING, - REPLACE_PAIR_COALESCED_WARNING, -} from "./constants"; +/** + * Token-driven state machine that turns a stream of {@link Token}s into a + * flat list of {@link Edit}s. Sits between the {@link Tokenizer} and the + * applier. + * + * Lifecycle: + * + * 1. Construct one {@link Executor} per hunk (or share one with `reset()`). + * 2. Feed it tokens via {@link Executor.feed}. Multi-line payloads are + * accumulated across tokens until the next op flushes them. + * 3. Call {@link Executor.end} to flush the trailing pending op and validate + * cross-op invariants (no overlapping deletes, etc.). + * + * Convenience entry point: {@link parsePatch}. + */ import { HL_OP_CHARS, HL_OP_DELETE, @@ -12,15 +20,16 @@ import { HL_OP_INSERT_BEFORE, HL_OP_REPLACE, HL_PAYLOAD_PREFIX, -} from "./hash"; +} from "./format"; import { - cloneCursor, - type HashlineToken, - HashlineTokenizer, - isDeleteOpWithPayload, - type ParsedRange, -} from "./tokenizer"; -import type { Anchor, HashlineCursor, HashlineEdit } from "./types"; + ABORT_WARNING, + IMPLICIT_CONTINUATION_WARNING, + INLINE_PAYLOAD_ACCEPTED_WARNING, + PAYLOAD_LINE_PREFIX_DEMOTED_WARNING, + REPLACE_PAIR_COALESCED_WARNING, +} from "./messages"; +import { cloneCursor, isDeleteOpWithPayload, type ParsedRange, type Token, Tokenizer } from "./tokenizer"; +import type { Anchor, Cursor, Edit } from "./types"; function validateRangeOrder(range: ParsedRange, lineNum: number): void { if (range.end.line < range.start.line) { @@ -45,7 +54,7 @@ function expandRange(range: ParsedRange): Anchor[] { } type PendingOp = - | { kind: "insert"; cursor: HashlineCursor; lineNum: number } + | { kind: "insert"; cursor: Cursor; lineNum: number } | { kind: "replace"; range: ParsedRange; lineNum: number }; interface Pending { @@ -54,24 +63,16 @@ interface Pending { } /** - * Token-driven state machine that turns a stream of {@link HashlineToken}s - * into the flat list of {@link HashlineEdit}s applied downstream by the - * apply/diff layers. + * Token-driven state machine that turns a stream of {@link Token}s into a + * flat list of {@link Edit}s. * - * The executor owns: - * - the running edit index (kept monotonic across pending flushes), - * - the pending-payload buffer (lines accumulated for the most recently - * opened insert/replace op), - * - all parse-time diagnostics (range order, "delete with payload", - * orphan payload, unrecognized op), - * - the {@link terminated} flag set by `envelope-end`/`abort`. - * - * Tokens are dispatched in the order they arrive; the matching tokenizer - * supplies the line numbers carried inside each token so diagnostics line - * up with the source. + * `feed()` accepts tokens one at a time; multi-line payloads accumulate + * until the next op or {@link end} flushes them. After `terminated` flips + * true (on `envelope-end` or `abort`) subsequent feeds are silently ignored + * so callers can keep draining their tokenizer. */ -export class HashlineExecutor { - #edits: HashlineEdit[] = []; +export class Executor { + #edits: Edit[] = []; #warnings: string[] = []; #editIndex = 0; #pending: Pending | undefined; @@ -83,11 +84,11 @@ export class HashlineExecutor { } /** - * Consume one token. After `terminated` flips true subsequent feeds - * are silently ignored so callers can keep draining their tokenizer - * without explicit early-exit guards. + * Consume one token. After `terminated` flips true subsequent feeds are + * silently ignored so callers can keep draining their tokenizer without + * explicit early-exit guards. */ - feed(token: HashlineToken): void { + feed(token: Token): void { if (this.#terminated) return; switch (token.kind) { @@ -182,14 +183,16 @@ export class HashlineExecutor { /** * Flush any open pending op (with its full accumulated payload, including - * explicit `+` blank lines) and return the accumulated edits and warnings. - * The executor is single-use; reset() is required for reuse. + * explicit `+` blank lines) and return the accumulated edits and + * warnings. The executor is single-use; {@link reset} is required for + * reuse. + * * Throws if two replace/delete ops target the same line with non-identical * shapes (different ranges, replace+delete, delete+delete). Identical-range * `A-B:` pairs in the same hunk are coalesced last-wins by `feed()` with a * warning, so they never reach the validator. */ - end(): { edits: HashlineEdit[]; warnings: string[] } { + end(): { edits: Edit[]; warnings: string[] } { this.#flushPending(); this.#validateNoOverlappingDeletes(); return { edits: this.#edits, warnings: this.#warnings }; @@ -253,7 +256,7 @@ export class HashlineExecutor { // Lenient legacy fallback: the tokenizer routes a line to `raw` only // when it does not parse as an op, header, payload, or envelope // marker. A `raw` token while a pending op exists is therefore an - // unambiguous continuation row that the model authored without the + // unambiguous continuation row that the author wrote without the // `+` prefix. Accept it as payload and warn so the canonical // `+`-prefixed form remains preferred. this.#pending.payload.push(text); @@ -267,7 +270,7 @@ export class HashlineExecutor { // fully empty lines arrive as `blank` tokens. if (text.trim().length === 0) return; // Orphan raw text outside any pending op: pick the most specific - // diagnostic so the model sees the actionable hint. + // diagnostic so the user sees the actionable hint. if (isDeleteOpWithPayload(text)) { throw new Error( `line ${lineNum}: ${HL_OP_DELETE} deletes only. Payload is forbidden after ${HL_OP_DELETE}; use ${HL_OP_REPLACE} to replace.`, @@ -328,14 +331,14 @@ export class HashlineExecutor { /** * Drive a full hashline diff through the tokenizer + executor pipeline and * return the resulting edits plus any parse-time warnings. This is the - * convenience entry point most callers want; reach for {@link - * HashlineTokenizer}/{@link HashlineExecutor} directly only when you need - * streaming feeds, cross-section state, or custom token handling. + * convenience entry point most callers want; reach for {@link Tokenizer} / + * {@link Executor} directly only when you need streaming feeds, cross-section + * state, or custom token handling. */ -export function parseHashline(diff: string): { edits: HashlineEdit[]; warnings: string[] } { - const tokenizer = new HashlineTokenizer(); - const executor = new HashlineExecutor(); - const drain = (tokens: HashlineToken[]): void => { +export function parsePatch(diff: string): { edits: Edit[]; warnings: string[] } { + const tokenizer = new Tokenizer(); + const executor = new Executor(); + const drain = (tokens: Token[]): void => { for (const token of tokens) { if (executor.terminated) return; executor.feed(token); diff --git a/packages/hashline/src/patcher.ts b/packages/hashline/src/patcher.ts new file mode 100644 index 000000000..9c4d4e482 --- /dev/null +++ b/packages/hashline/src/patcher.ts @@ -0,0 +1,343 @@ +/** + * High-level patch orchestrator. Reads each section's target file via the + * configured {@link Filesystem}, strips BOM and normalizes line endings, + * validates the section file hash (with optional {@link Recovery}), applies + * the edits, and writes the result back through the same {@link Filesystem}. + * + * Two layers: + * + * - {@link Patcher.apply} — high-level, all-or-nothing. Preflights every + * section in memory before any write hits disk, then commits in order. + * - {@link Patcher.prepare} / {@link Patcher.commit} — granular primitives + * for callers that need per-section control (e.g. batched LSP flush, + * custom interleaving). `prepare` performs all the read-side work, + * validates the section file hash (with recovery), and applies the + * edits in memory. `commit` writes the prepared result and records a + * fresh snapshot. + * + * Because `prepare` already runs the full apply, a multi-section batch is + * naturally all-or-nothing: by the time any `commit` runs, every section + * has been validated. + * + * The patcher itself is stateless across calls; reuse one instance per + * filesystem configuration. + */ +import { applyEdits } from "./apply"; +import { computeFileHash, formatHashlineHeader, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import type { Filesystem, WriteResult } from "./fs"; +import { isNotFound } from "./fs"; +import type { Patch, PatchSection } from "./input"; +import { MismatchError } from "./mismatch"; +import { detectLineEnding, type LineEnding, normalizeToLF, restoreLineEndings, stripBom } from "./normalize"; +import { Recovery, type RecoveryResult } from "./recovery"; +import type { SnapshotStore } from "./snapshots"; +import type { ApplyOptions, ApplyResult, Edit } from "./types"; + +export interface PatcherOptions { + /** Storage backend used for all reads and writes. */ + fs: Filesystem; + /** + * Optional snapshot store that enables stale-hash recovery. When set, a + * section with a stale hash tries a 3-way merge against a cached + * snapshot before the apply fails with {@link MismatchError}. + */ + snapshots?: SnapshotStore; + /** + * Optional default {@link ApplyOptions} forwarded to every section. + * Per-call overrides win on a key-by-key basis. + */ + applyOptions?: ApplyOptions; +} + +/** Per-section result returned by {@link Patcher.apply} / {@link Patcher.commit}. */ +export interface PatchSectionResult { + /** Section path (as authored, after cwd-resolution at parse time). */ + path: string; + /** Filesystem-canonical key for this section (e.g. absolute path). */ + canonicalPath: string; + /** `"noop"` when the apply produced no change; otherwise `"create"` / `"update"`. */ + op: "create" | "update" | "noop"; + /** Pre-edit text (LF-normalized, BOM-stripped). */ + before: string; + /** Post-edit text (LF-normalized, BOM-stripped). For `"noop"` equals `before`. */ + after: string; + /** Same text as `after` but with the original BOM and line ending restored. */ + persisted: string; + /** Final text that the {@link Filesystem} actually wrote (may differ if the FS transformed it). */ + written: string; + /** 4-hex hash of `after`. Use to anchor follow-up edits. */ + fileHash: string; + /** Hashline section header (`¶path#hash`) of the post-edit content. */ + header: string; + /** 1-indexed first changed line in `after`, or `undefined` for noops. */ + firstChangedLine?: number; + /** Warnings collected by the parser, applier, and (optionally) recovery. */ + warnings: string[]; +} + +export interface PatcherApplyResult { + sections: PatchSectionResult[]; +} + +/** + * Opaque token returned by {@link Patcher.prepare}. Carries the section, the + * raw file content read off disk, and the in-memory apply result. + * {@link Patcher.commit} just writes the {@link PreparedSection.applyResult}. + */ +export class PreparedSection { + /** @internal */ + constructor( + readonly section: PatchSection, + readonly canonicalPath: string, + readonly exists: boolean, + readonly rawContent: string, + readonly bom: string, + readonly lineEnding: LineEnding, + readonly normalized: string, + readonly applyResult: ApplyResult, + readonly parseWarnings: readonly string[], + ) {} + + /** Convenience: returns true when the apply produced no change. */ + get isNoop(): boolean { + return this.applyResult.text === this.normalized; + } +} + +function hasAnchorScopedEdit(edits: readonly Edit[]): boolean { + return edits.some(edit => { + if (edit.kind === "delete") return true; + return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor"; + }); +} + +function assertSectionHashAllowed(sectionPath: string, fileHash: string | undefined, edits: readonly Edit[]): void { + if (fileHash !== undefined || !hasAnchorScopedEdit(edits)) return; + throw new Error( + `Missing hashline file hash for anchored edit to ${sectionPath}; use \`${HL_FILE_PREFIX}${sectionPath}${HL_FILE_HASH_SEP}hash\` from your latest read.`, + ); +} + +function recoveryToApplyResult(result: RecoveryResult): ApplyResult { + return { + text: result.text, + firstChangedLine: result.firstChangedLine, + warnings: result.warnings, + }; +} + +function mergeWarnings(...sources: ReadonlyArray): string[] { + const out: string[] = []; + for (const source of sources) { + if (!source) continue; + for (const warning of source) out.push(warning); + } + return out; +} + +/** + * High-level patcher. Wires a {@link Filesystem} and an optional + * {@link SnapshotStore} together with the parsing + applying core. + * + * Construct once per FS configuration; reuse across patches. + */ +export class Patcher { + readonly fs: Filesystem; + readonly snapshots: SnapshotStore | undefined; + readonly recovery: Recovery | undefined; + readonly applyOptions: ApplyOptions; + + constructor(options: PatcherOptions) { + this.fs = options.fs; + this.snapshots = options.snapshots; + this.recovery = options.snapshots ? new Recovery(options.snapshots) : undefined; + this.applyOptions = options.applyOptions ?? {}; + } + + /** + * Apply every section in `patch`. `prepare` runs the full apply for each + * section in memory before any write hits the filesystem, so a + * multi-section batch is naturally all-or-nothing. Returns one + * {@link PatchSectionResult} per section in the original patch order. + */ + async apply(patch: Patch, options: ApplyOptions = {}): Promise { + const merged: ApplyOptions = { ...this.applyOptions, ...options }; + + // Single-section fast path. + if (patch.sections.length === 1) { + const prepared = await this.prepare(patch.sections[0], merged); + return { sections: [await this.commit(prepared)] }; + } + + // Prepare every section first so any failure (stale hash, missing + // file, parse error, in-memory no-op) surfaces before any write. + const prepared: PreparedSection[] = []; + for (const section of patch.sections) prepared.push(await this.prepare(section, merged)); + for (const entry of prepared) { + if (entry.isNoop) { + throw new Error(`Edits to ${entry.section.path} resulted in no changes being made.`); + } + } + + const results: PatchSectionResult[] = []; + for (const entry of prepared) results.push(await this.commit(entry)); + return { sections: results }; + } + + /** + * Run the preflight pass only: read, parse, validate, apply-in-memory. + * No writes hit the filesystem. Use for CI checks and dry runs. + */ + async preflight(patch: Patch, options: ApplyOptions = {}): Promise { + const merged: ApplyOptions = { ...this.applyOptions, ...options }; + for (const section of patch.sections) { + const prepared = await this.prepare(section, merged); + if (prepared.isNoop) { + throw new Error(`Edits to ${section.path} resulted in no changes being made.`); + } + } + } + + /** + * Read a section's target file, parse the section, validate the file + * hash (with recovery), and apply the edits in memory. Returns a + * {@link PreparedSection} which can be fed to {@link commit} to land + * the result on the filesystem. + * + * Throws on parse error, missing-file-for-anchored-edit, or unrecovered + * hash mismatch ({@link MismatchError}). + */ + async prepare(section: PatchSection, options: ApplyOptions = {}): Promise { + const applyOptions: ApplyOptions = { ...this.applyOptions, ...options }; + const { edits, warnings: parseWarnings } = section.parse(); + assertSectionHashAllowed(section.path, section.fileHash, edits); + + const canonicalPath = this.fs.canonicalPath(section.path); + const { exists, rawContent } = await this.#tryRead(section.path); + if (!exists && hasAnchorScopedEdit(edits)) { + throw new Error(`File not found: ${section.path}`); + } + + const { bom, text } = stripBom(rawContent); + const lineEnding = detectLineEnding(text); + const normalized = normalizeToLF(text); + + const applyResult = this.#applyWithRecovery({ + section, + canonicalPath, + exists, + normalized, + edits, + applyOptions, + }); + + return new PreparedSection( + section, + canonicalPath, + exists, + rawContent, + bom, + lineEnding, + normalized, + applyResult, + parseWarnings, + ); + } + + /** + * Commit a previously {@link prepare}d section to the filesystem. + * Restores line endings and BOM, writes via the {@link Filesystem}, and + * records a fresh snapshot in the {@link SnapshotStore} (when + * configured) keyed by the filesystem-canonical path. + */ + async commit(prepared: PreparedSection): Promise { + const { section, normalized, bom, lineEnding, parseWarnings, exists, applyResult, canonicalPath } = prepared; + const after = applyResult.text; + const warnings = mergeWarnings(parseWarnings, applyResult.warnings); + + if (after === normalized) { + const hash = computeFileHash(normalized); + return { + path: section.path, + canonicalPath, + op: "noop", + before: normalized, + after: normalized, + persisted: prepared.rawContent, + written: prepared.rawContent, + fileHash: hash, + header: formatHashlineHeader(section.path, hash), + warnings, + }; + } + + const persisted = bom + restoreLineEndings(after, lineEnding); + const write: WriteResult = await this.fs.writeText(section.path, persisted); + const fileHash = computeFileHash(after); + const op = exists ? "update" : "create"; + + if (this.snapshots) { + this.snapshots.recordContiguous(canonicalPath, 1, after.split("\n"), { + fullText: after, + fileHash, + }); + } + + return { + path: section.path, + canonicalPath, + op, + before: normalized, + after, + persisted, + written: write.text, + fileHash, + header: formatHashlineHeader(section.path, fileHash), + firstChangedLine: applyResult.firstChangedLine, + warnings, + }; + } + + async #tryRead(path: string): Promise<{ exists: boolean; rawContent: string }> { + try { + const content = await this.fs.readText(path); + return { exists: true, rawContent: content }; + } catch (error) { + if (isNotFound(error)) return { exists: false, rawContent: "" }; + throw error; + } + } + + #applyWithRecovery(args: { + section: PatchSection; + canonicalPath: string; + exists: boolean; + normalized: string; + edits: readonly Edit[]; + applyOptions: ApplyOptions; + }): ApplyResult { + const { section, canonicalPath, exists, normalized, edits, applyOptions } = args; + const expected = exists ? section.fileHash : undefined; + if (expected === undefined) return applyEdits(normalized, [...edits], applyOptions); + + const currentHash = computeFileHash(normalized); + if (currentHash === expected) return applyEdits(normalized, [...edits], applyOptions); + + const recovered = this.recovery?.tryRecover({ + path: canonicalPath, + currentText: normalized, + fileHash: expected, + edits, + options: applyOptions, + }); + if (recovered) return recoveryToApplyResult(recovered); + + throw new MismatchError({ + path: section.path, + expectedFileHash: expected, + actualFileHash: currentHash, + fileLines: normalized.split("\n"), + anchorLines: section.collectAnchorLines(), + }); + } +} diff --git a/packages/coding-agent/src/hashline/prefixes.ts b/packages/hashline/src/prefixes.ts similarity index 72% rename from packages/coding-agent/src/hashline/prefixes.ts rename to packages/hashline/src/prefixes.ts index 057870154..678c4625f 100644 --- a/packages/coding-agent/src/hashline/prefixes.ts +++ b/packages/hashline/src/prefixes.ts @@ -1,3 +1,19 @@ +/** + * When a hashline payload is authored against `read`/`search` output, each + * line is prefixed with either a hashline-mode line number (`123:`) or, for + * diff-style echoes, a leading `+`. These helpers detect that and recover + * the raw text. Two strip modes are exposed: + * + * - {@link stripNewLinePrefixes} — opportunistic: strips when the input + * clearly carries hashline or diff prefixes, leaves it alone otherwise. + * - {@link stripHashlinePrefixes} — strict: only strips when every non-empty + * content line is hashline-prefixed. + * + * These run *before* the tokenizer; they exist because hashline mode is the + * common case for echoed file content, and erroneously echoed prefixes will + * otherwise turn every content line into a (malformed) op. + */ + const HL_PREFIX_RE = /^\s*(?:>>>|>>)?\s*(?:[+*-]\s*)?\d+:/; const HL_PREFIX_PLUS_RE = /^\s*(?:>>>|>>)?\s*\+\s*\d+:/; const HL_HEADER_RE = /^\s*¶\S+#[0-9a-f]{4}\s*$/; @@ -14,23 +30,14 @@ function stripLeadingHashlinePrefixes(line: string): string { return result; } -// ─────────────────────────────────────────────────────────────────────────── -// 5. Read-output prefix stripping -// -// When a model echoes back content from a `read` or `search` response, every -// line is prefixed with either a hashline-mode line number (`123:`) or, for -// diff-style echoes, a leading `+`. These helpers detect that and recover the -// raw text. -// ─────────────────────────────────────────────────────────────────────────── - -type LinePrefixStats = { +interface LinePrefixStats { nonEmpty: number; headerCount: number; hashPrefixCount: number; diffPlusHashPrefixCount: number; diffPlusCount: number; truncationNoticeCount: number; -}; +} function collectLinePrefixStats(lines: string[]): LinePrefixStats { const stats: LinePrefixStats = { @@ -61,6 +68,14 @@ function collectLinePrefixStats(lines: string[]): LinePrefixStats { return stats; } +/** + * Strip whichever prefix scheme the lines appear to be carrying: + * - hashline line-number prefixes (`123:`) when every content line has one + * - leading `+` (diff style) when at least half the lines have one + * - mixed `+:` form when present + * + * Returns the lines untouched if no scheme is recognized. + */ export function stripNewLinePrefixes(lines: string[]): string[] { const stats = collectLinePrefixStats(lines); if (stats.nonEmpty === 0) return lines; @@ -87,6 +102,10 @@ export function stripNewLinePrefixes(lines: string[]): string[] { }); } +/** + * Strict variant: strip hashline prefixes only when every content line is + * hashline-prefixed. Returns the lines unchanged otherwise. + */ export function stripHashlinePrefixes(lines: string[]): string[] { const stats = collectLinePrefixStats(lines); if (stats.nonEmpty === 0) return lines; diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/hashline/src/prompt.md similarity index 100% rename from packages/coding-agent/src/prompts/tools/hashline.md rename to packages/hashline/src/prompt.md diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts new file mode 100644 index 000000000..498ae358d --- /dev/null +++ b/packages/hashline/src/recovery.ts @@ -0,0 +1,163 @@ +/** + * Recover from a stale section file-hash by replaying the would-be edit + * against a cached pre-edit snapshot of the file and 3-way-merging the + * result onto the current on-disk content. + * + * The patcher consults this when it sees a section hash that doesn't match + * the live file content. The recovery class is stateless apart from the + * {@link SnapshotStore} it queries; the snapshot store is the seam that + * lets you plug in your own caching strategy. + */ +import * as Diff from "diff"; +import { applyEdits } from "./apply"; +import { computeFileHash } from "./format"; +import { RECOVERY_EXTERNAL_WARNING, RECOVERY_SESSION_CHAIN_WARNING, RECOVERY_SESSION_REPLAY_WARNING } from "./messages"; +import type { Snapshot, SnapshotStore } from "./snapshots"; +import type { ApplyOptions, ApplyResult, Edit } from "./types"; + +// Section hashes are line-precise; never let Diff.applyPatch slide a hunk +// onto a duplicate closer 100+ lines away. If snapshot replay does not +// align exactly, refuse and let the caller re-read. +const RECOVERY_FUZZ_FACTOR = 0; + +export interface RecoveryArgs { + path: string; + currentText: string; + fileHash: string; + edits: readonly Edit[]; + options?: ApplyOptions; +} + +export interface RecoveryResult { + /** Post-recovery text. */ + text: string; + /** First changed line (1-indexed) relative to the live `currentText`, or `undefined`. */ + firstChangedLine: number | undefined; + /** Warnings collected during recovery, including the user-facing recovery banner. */ + warnings: string[]; +} + +function applyEditsToSnapshot( + previousText: string, + currentText: string, + edits: readonly Edit[], + options: ApplyOptions, + recoveryWarning: string, +): RecoveryResult | null { + let applied: ApplyResult; + try { + applied = applyEdits(previousText, [...edits], options); + } catch { + return null; + } + if (applied.text === previousText) return null; + + const patch = Diff.structuredPatch("file", "file", previousText, applied.text, "", "", { context: 3 }); + const merged = Diff.applyPatch(currentText, patch, { fuzzFactor: RECOVERY_FUZZ_FACTOR }); + if (typeof merged !== "string" || merged === currentText) return null; + + const firstChangedLine = findFirstChangedLine(currentText, merged) ?? applied.firstChangedLine; + const hasNetChange = firstChangedLine !== undefined; + const warnings = hasNetChange ? [recoveryWarning, ...(applied.warnings ?? [])] : [...(applied.warnings ?? [])]; + + return { text: merged, firstChangedLine, warnings }; +} + +function replaySessionChainOnCurrent( + previousText: string, + currentText: string, + edits: readonly Edit[], + options: ApplyOptions, +): RecoveryResult | null { + // Only safe when no insert/delete shifted line counts in the prior edit + // chain: if total line counts match, every line number in `edits` still + // resolves to the same logical row. + if (previousText.split("\n").length !== currentText.split("\n").length) return null; + let applied: ApplyResult; + try { + applied = applyEdits(currentText, [...edits], options); + } catch { + return null; + } + if (applied.text === currentText) return null; + return { + text: applied.text, + firstChangedLine: applied.firstChangedLine, + warnings: [RECOVERY_SESSION_REPLAY_WARNING, ...(applied.warnings ?? [])], + }; +} + +function buildSparseOverlayText(currentText: string, snapshotLines: ReadonlyMap): string { + const overlaid = currentText.split("\n"); + let maxCachedLine = 0; + for (const lineNum of snapshotLines.keys()) { + if (lineNum > maxCachedLine) maxCachedLine = lineNum; + } + while (overlaid.length < maxCachedLine) overlaid.push(""); + for (const [lineNum, content] of snapshotLines) { + overlaid[lineNum - 1] = content; + } + return overlaid.join("\n"); +} + +/** First 1-indexed line at which `a` and `b` diverge, or `undefined` if equal. */ +function findFirstChangedLine(a: string, b: string): number | undefined { + if (a === b) return undefined; + const aLines = a.split("\n"); + const bLines = b.split("\n"); + const max = Math.max(aLines.length, bLines.length); + for (let i = 0; i < max; i++) { + if (aLines[i] !== bLines[i]) return i + 1; + } + return undefined; +} + +function isHeadSnapshot(head: Snapshot | null, snapshot: Snapshot): boolean { + return head === snapshot; +} + +/** + * Stateless recovery driver over a {@link SnapshotStore}. Construct once and + * call {@link Recovery.tryRecover} per stale-hash incident. The default + * implementation tries three strategies in order: + * + * 1. Apply on the cached `fullText` snapshot, then 3-way-merge onto current. + * 2. (Session chain) If the snapshot wasn't the head, retry on current text + * when line counts match — the user's previous edit advanced the hash but + * didn't shift line numbers. + * 3. Reconstruct from a sparse snapshot (lines map only), verify the rebuilt + * text hashes to the expected value, then 3-way-merge. + */ +export class Recovery { + constructor(readonly store: SnapshotStore) {} + + /** + * Attempt recovery. Returns `null` when no path forward is found — the + * caller should then surface a {@link MismatchError}. + */ + tryRecover(args: RecoveryArgs): RecoveryResult | null { + const { path, currentText, fileHash, edits, options = {} } = args; + const head = this.store.head(path); + const snapshot = this.store.byHash(path, fileHash); + if (!snapshot || snapshot.lines.size === 0) return null; + + const isHead = isHeadSnapshot(head, snapshot); + const recoveryWarning = isHead ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; + const isSessionChain = !isHead; + + if (snapshot.fullText !== undefined) { + const merged = applyEditsToSnapshot(snapshot.fullText, currentText, edits, options, recoveryWarning); + if (merged !== null) return merged; + // Session-chain fast-path: prior in-session edit changed the same + // line(s) the user is now re-targeting with the stale hash. When + // line counts match, the edits' line numbers still resolve to the + // right rows — replay onto the current text directly. + if (isSessionChain) return replaySessionChainOnCurrent(snapshot.fullText, currentText, edits, options); + return null; + } + + const overlayText = buildSparseOverlayText(currentText, snapshot.lines); + if (computeFileHash(overlayText) !== fileHash) return null; + return applyEditsToSnapshot(overlayText, currentText, edits, options, recoveryWarning); + } +} diff --git a/packages/hashline/src/snapshots.ts b/packages/hashline/src/snapshots.ts new file mode 100644 index 000000000..213324f12 --- /dev/null +++ b/packages/hashline/src/snapshots.ts @@ -0,0 +1,171 @@ +/** + * Per-session snapshot store used by {@link Recovery} to rescue patches when + * a section's file hash has drifted (file changed externally or a prior + * in-session edit advanced the hash). + * + * Producers (typically a `read` tool) record snapshots as they observe file + * content. Consumers (the patcher) query for the snapshot whose hash matches + * a stale section, then 3-way-merge the would-be edit onto the live content. + * + * The abstract base class lets callers plug in whatever storage they like + * (LRU, persistent SQLite, etc.). {@link InMemorySnapshotStore} ships as a + * sensible default backed by `lru-cache` so paths age out automatically. + */ +import { LRUCache } from "lru-cache/raw"; + +/** + * One snapshot of a file as it was observed at a point in time. Either the + * full text is recorded (`fullText` set) for full-file reads, or a sparse + * map of `(lineNumber, content)` pairs for partial views (search matches, + * range reads). + */ +export interface Snapshot { + /** 1-indexed line number → exact content as observed. */ + readonly lines: Map; + /** Full normalized text when the read observed the whole file. */ + fullText?: string; + /** 4-hex hash carried alongside the read, when known. */ + fileHash?: string; + /** Timestamp (ms since epoch) the snapshot was recorded. */ + recordedAt: number; +} + +/** Optional metadata supplied at snapshot record time. */ +export interface SnapshotMetadata { + /** Full normalized text, when the producer observed the whole file. */ + fullText?: string; + /** 4-hex hash carried by the read, when known. */ + fileHash?: string; +} + +/** + * Storage seam for file-content snapshots. The patcher calls {@link head} + * for the latest snapshot of a path and {@link byHash} when it needs the + * specific historical snapshot that matches a section's stale hash. + */ +export abstract class SnapshotStore { + /** Most-recent snapshot for `path`, or `null` if none. */ + abstract head(path: string): Snapshot | null; + + /** Most-recent snapshot for `path` whose `fileHash` equals `fileHash`. */ + abstract byHash(path: string, fileHash: string): Snapshot | null; + + /** Record a contiguous run of lines (e.g. from a `read` tool). `startLine` is 1-indexed. */ + abstract recordContiguous( + path: string, + startLine: number, + lines: readonly string[], + metadata?: SnapshotMetadata, + ): void; + + /** Record sparse `(lineNumber, content)` pairs (e.g. a `search` match plus context). */ + abstract recordSparse(path: string, entries: Iterable, metadata?: SnapshotMetadata): void; + + /** Drop the snapshot history for a single path. */ + abstract invalidate(path: string): void; + + /** Drop every snapshot history. */ + abstract clear(): void; +} + +const DEFAULT_MAX_PATHS = 30; +const DEFAULT_MAX_SNAPSHOTS_PER_PATH = 4; + +function hasConflict( + existing: ReadonlyMap, + incoming: ReadonlyArray, +): boolean { + for (const [lineNum, content] of incoming) { + const prior = existing.get(lineNum); + if (prior !== undefined && prior !== content) return true; + } + return false; +} + +function hasHashConflict(existing: Snapshot, metadata: SnapshotMetadata): boolean { + return metadata.fileHash !== undefined && existing.fileHash !== undefined && metadata.fileHash !== existing.fileHash; +} + +function isSameSnapshotIdentity(left: Snapshot, right: Snapshot): boolean { + if (left.fileHash !== undefined && right.fileHash !== undefined) return left.fileHash === right.fileHash; + if (left.fullText !== undefined && right.fullText !== undefined) return left.fullText === right.fullText; + return false; +} + +export interface InMemorySnapshotStoreOptions { + /** Maximum number of distinct paths tracked at once (default 30). LRU eviction. */ + maxPaths?: number; + /** Maximum snapshots retained per path (default 4). Oldest dropped first. */ + maxSnapshotsPerPath?: number; +} + +/** + * In-memory {@link SnapshotStore} backed by `lru-cache`. Per-path snapshot + * history is a short ring (oldest dropped first); per-session path tracking + * is LRU-bounded so cold paths age out automatically. + * + * Newer snapshots merge into the head when their entries don't conflict and + * the recorded `fileHash` (if any) still agrees; otherwise a fresh snapshot + * is pushed onto the front of the history list. + */ +export class InMemorySnapshotStore extends SnapshotStore { + readonly #snapshots: LRUCache; + readonly #maxSnapshotsPerPath: number; + + constructor(options: InMemorySnapshotStoreOptions = {}) { + super(); + this.#snapshots = new LRUCache({ max: options.maxPaths ?? DEFAULT_MAX_PATHS }); + this.#maxSnapshotsPerPath = options.maxSnapshotsPerPath ?? DEFAULT_MAX_SNAPSHOTS_PER_PATH; + } + + head(path: string): Snapshot | null { + return this.#snapshots.get(path)?.[0] ?? null; + } + + byHash(path: string, fileHash: string): Snapshot | null { + const history = this.#snapshots.get(path); + return history?.find(entry => entry.fileHash === fileHash) ?? null; + } + + recordContiguous(path: string, startLine: number, lines: readonly string[], metadata: SnapshotMetadata = {}): void { + if (lines.length === 0 && metadata.fullText === undefined) return; + const entries: Array = lines.map((line, idx) => [startLine + idx, line] as const); + this.#record(path, entries, metadata); + } + + recordSparse(path: string, entries: Iterable, metadata: SnapshotMetadata = {}): void { + const arr = Array.from(entries); + if (arr.length === 0 && metadata.fullText === undefined) return; + this.#record(path, arr, metadata); + } + + invalidate(path: string): void { + this.#snapshots.delete(path); + } + + clear(): void { + this.#snapshots.clear(); + } + + #record(path: string, entries: ReadonlyArray, metadata: SnapshotMetadata): void { + const history = this.#snapshots.get(path) ?? []; + const head = history[0]; + const now = Date.now(); + if (head && !hasConflict(head.lines, entries) && !hasHashConflict(head, metadata)) { + for (const [lineNum, content] of entries) head.lines.set(lineNum, content); + if (metadata.fullText !== undefined) head.fullText = metadata.fullText; + if (metadata.fileHash !== undefined) head.fileHash = metadata.fileHash; + head.recordedAt = now; + // `get` above already touched LRU recency for this key. + return; + } + + const nextSnapshot: Snapshot = { + lines: new Map(entries), + ...metadata, + recordedAt: now, + }; + const deduped = history.filter(entry => !isSameSnapshotIdentity(entry, nextSnapshot)); + this.#snapshots.set(path, [nextSnapshot, ...deduped].slice(0, this.#maxSnapshotsPerPath)); + } +} diff --git a/packages/coding-agent/src/hashline/stream.ts b/packages/hashline/src/stream.ts similarity index 77% rename from packages/coding-agent/src/hashline/stream.ts rename to packages/hashline/src/stream.ts index 05b4bd2b4..337c1df51 100644 --- a/packages/coding-agent/src/hashline/stream.ts +++ b/packages/hashline/src/stream.ts @@ -1,13 +1,22 @@ -import { formatNumberedLine } from "./hash"; -import type { HashlineStreamOptions } from "./types"; +/** + * Lazily format a stream of UTF-8 bytes into hashline-numbered lines, yielded + * as bounded text chunks. Used to send `read`-style file content to consumers + * without materializing the full file at once. + * + * Each yielded chunk is at most {@link StreamOptions.maxChunkLines} lines and + * at most {@link StreamOptions.maxChunkBytes} UTF-8 bytes (whichever fires + * first). + */ +import { formatNumberedLine } from "./format"; +import type { StreamOptions } from "./types"; -interface ResolvedHashlineStreamOptions { +interface ResolvedStreamOptions { startLine: number; maxChunkLines: number; maxChunkBytes: number; } -function resolveHashlineStreamOptions(options: HashlineStreamOptions): ResolvedHashlineStreamOptions { +function resolveStreamOptions(options: StreamOptions): ResolvedStreamOptions { return { startLine: options.startLine ?? 1, maxChunkLines: options.maxChunkLines ?? 200, @@ -15,12 +24,12 @@ function resolveHashlineStreamOptions(options: HashlineStreamOptions): ResolvedH }; } -interface HashlineChunkEmitter { +interface ChunkEmitter { pushLine: (line: string) => string[]; flush: () => string | undefined; } -function createHashlineChunkEmitter(options: ResolvedHashlineStreamOptions): HashlineChunkEmitter { +function createChunkEmitter(options: ResolvedStreamOptions): ChunkEmitter { let lineNumber = options.startLine; let outLines: string[] = []; let outBytes = 0; @@ -83,14 +92,14 @@ async function* bytesFromReadableStream(stream: ReadableStream): Asy } } -export async function* streamHashLinesFromUtf8( +export async function* streamHashLines( source: ReadableStream | AsyncIterable, - options: HashlineStreamOptions = {}, + options: StreamOptions = {}, ): AsyncGenerator { - const resolved = resolveHashlineStreamOptions(options); + const resolved = resolveStreamOptions(options); const decoder = new TextDecoder("utf-8"); const chunks = isReadableStream(source) ? bytesFromReadableStream(source) : source; - const emitter = createHashlineChunkEmitter(resolved); + const emitter = createChunkEmitter(resolved); let pending = ""; let sawAnyLine = false; diff --git a/packages/coding-agent/src/hashline/tokenizer.ts b/packages/hashline/src/tokenizer.ts similarity index 86% rename from packages/coding-agent/src/hashline/tokenizer.ts rename to packages/hashline/src/tokenizer.ts index cc47ff8a3..f50471061 100644 --- a/packages/coding-agent/src/hashline/tokenizer.ts +++ b/packages/hashline/src/tokenizer.ts @@ -1,4 +1,18 @@ -import { ABORT_MARKER, BEGIN_PATCH_MARKER, END_PATCH_MARKER } from "./constants"; +/** + * Stateful, line-oriented classifier for hashline diff text. + * + * The {@link Tokenizer} can be fed in chunks ({@link Tokenizer.feed}/{@link + * Tokenizer.end}) for streaming use, or in one shot ({@link + * Tokenizer.tokenizeAll}). Each emitted token carries its 1-indexed source + * line number so downstream consumers (parser, validators, error messages) + * can refer back to the input precisely. + * + * The tokenizer is intentionally permissive about decorations and prefixes + * the model may echo back from `read`/`search` output — leading `*`/`>`/`-` + * markers, CR-terminated lines, leading whitespace before line numbers, and + * so on are all stripped before classification. + */ + import { describeAnchorExamples, HL_FILE_HASH_SEP, @@ -8,8 +22,9 @@ import { HL_OP_INSERT_BEFORE, HL_OP_REPLACE, HL_PAYLOAD_PREFIX, -} from "./hash"; -import type { Anchor, HashlineCursor } from "./types"; +} from "./format"; +import { ABORT_MARKER, BEGIN_PATCH_MARKER, END_PATCH_MARKER } from "./messages"; +import type { Anchor, Cursor, ParsedRange } from "./types"; const CHAR_LINE_FEED = 10; const CHAR_CARRIAGE_RETURN = 13; @@ -61,9 +76,9 @@ function markerLineEquals(line: string, marker: string): boolean { * empty line that callers may rely on for explicit blank payloads. CRLF pairs * are normalized to a single line break. * - * This mirrors the line-splitting performed by {@link HashlineTokenizer}'s - * streaming drain loop and is kept for non-streaming callers that prefer - * a single-shot split. + * This mirrors the line-splitting performed by {@link Tokenizer}'s streaming + * drain loop and is kept for non-streaming callers that prefer a single-shot + * split. */ export function splitHashlineLines(text: string): string[] { if (text.length === 0) return [""]; @@ -86,7 +101,7 @@ export function splitHashlineLines(text: string): string[] { return lines; } -export function cloneCursor(cursor: HashlineCursor): HashlineCursor { +export function cloneCursor(cursor: Cursor): Cursor { if (cursor.kind === "before_anchor") return { kind: "before_anchor", anchor: { ...cursor.anchor } }; if (cursor.kind === "after_anchor") return { kind: "after_anchor", anchor: { ...cursor.anchor } }; return cursor; @@ -134,11 +149,6 @@ export function parseLid(raw: string, lineNum: number): Anchor { return { line: number.line }; } -export interface ParsedRange { - start: Anchor; - end: Anchor; -} - interface RangeScan { range: ParsedRange; nextIndex: number; @@ -172,7 +182,7 @@ function startsWithWord(line: string, index: number, end: number, word: string): return true; } -function parseInsertTarget(raw: string, lineNum: number, kind: "before" | "after"): HashlineCursor { +function parseInsertTarget(raw: string, lineNum: number, kind: "before" | "after"): Cursor { const end = trimEndIndex(raw); const targetStart = skipDecoratedAnchorPrefix(raw, end); @@ -194,7 +204,7 @@ function scanInlineBody(line: string, index: number): string | undefined { interface ParsedInsertOp { kind: "insert"; - cursor: HashlineCursor; + cursor: Cursor; inlineBody: string | undefined; } @@ -272,9 +282,10 @@ function tryParseOp(line: string): ParsedOp | null { } /** - * Strict header scan: `¶+` prefix, optional whitespace, path body that excludes - * whitespace, `#`, and `¶`, optional `#[0-9a-f]{4}` hash suffix, optional - * trailing whitespace. Returns `null` when any byte deviates from the shape. + * Strict header scan: `¶+` prefix, optional whitespace, path body that + * excludes whitespace, `#`, and `¶`, optional `#[0-9a-f]{4}` hash suffix, + * optional trailing whitespace. Returns `null` when any byte deviates from + * the shape. */ function tryParseHeader(line: string): { path: string; fileHash?: string } | null { const end = trimEndIndex(line); @@ -313,9 +324,9 @@ function tryParseHeader(line: string): { path: string; fileHash?: string } | nul } /** - * Returns true when the line scans as `LINE!payload` (delete sigil followed by - * additional content). The executor uses this for the dedicated "deletes only" - * diagnostic, separate from the standard "unrecognized op" path. + * Returns true when the line scans as `LINE!payload` (delete sigil followed + * by additional content). The parser uses this for the dedicated "deletes + * only" diagnostic, separate from the standard "unrecognized op" path. */ export function isDeleteOpWithPayload(line: string): boolean { const range = scanRange(line, line.length); @@ -332,19 +343,19 @@ interface TokenBase { lineNum: number; } -export type HashlineToken = +export type Token = | (TokenBase & { kind: "blank" }) | (TokenBase & { kind: "envelope-begin" }) | (TokenBase & { kind: "envelope-end" }) | (TokenBase & { kind: "abort" }) | (TokenBase & { kind: "header"; path: string; fileHash?: string }) - | (TokenBase & { kind: "op-insert"; cursor: HashlineCursor; inlineBody: string | undefined }) + | (TokenBase & { kind: "op-insert"; cursor: Cursor; inlineBody: string | undefined }) | (TokenBase & { kind: "op-replace"; range: ParsedRange; inlineBody: string | undefined }) | (TokenBase & { kind: "op-delete"; range: ParsedRange; trailingPayload: boolean }) | (TokenBase & { kind: "payload"; text: string }) | (TokenBase & { kind: "raw"; text: string }); -function classifyLine(line: string, lineNum: number): HashlineToken { +function classifyLine(line: string, lineNum: number): Token { if (isEmptyLine(line)) return { kind: "blank", lineNum }; if (markerLineEquals(line, BEGIN_PATCH_MARKER)) return { kind: "envelope-begin", lineNum }; if (markerLineEquals(line, END_PATCH_MARKER)) return { kind: "envelope-end", lineNum }; @@ -377,14 +388,14 @@ function classifyLine(line: string, lineNum: number): HashlineToken { } /** - * Stateful, line-oriented classifier for hashline diff text. Use the streaming - * {@link feed}/{@link end} pair to ingest text in chunks (each completed line - * emits exactly one token; a trailing partial line stays buffered until the - * next chunk or {@link end}). Use the stateless {@link tokenize}/predicate - * methods for callers that already hold whole lines and only need - * classification without buffering. + * Stateful, line-oriented classifier for hashline diff text. Use the + * streaming {@link feed}/{@link end} pair to ingest text in chunks (each + * completed line emits exactly one token; a trailing partial line stays + * buffered until the next chunk or {@link end}). Use the stateless + * {@link tokenize}/predicate methods for callers that already hold whole + * lines and only need classification without buffering. */ -export class HashlineTokenizer { +export class Tokenizer { #buffer = ""; #nextLineNum = 1; #closed = false; @@ -396,8 +407,8 @@ export class HashlineTokenizer { * `feed`/`end` call so CRLF pairs that straddle chunk boundaries are * still normalized correctly. */ - feed(chunk: string): HashlineToken[] { - if (this.#closed) throw new Error("HashlineTokenizer is closed; call reset() before reusing."); + feed(chunk: string): Token[] { + if (this.#closed) throw new Error("Tokenizer is closed; call reset() before reusing."); if (chunk.length === 0) return []; this.#buffer = this.#buffer ? this.#buffer + chunk : chunk; return this.#drainCompleteLines(); @@ -408,7 +419,7 @@ export class HashlineTokenizer { * a trailing newline) and mark the tokenizer closed. Calling `end` a * second time returns `[]`; reuse requires `reset`. */ - end(): HashlineToken[] { + end(): Token[] { if (this.#closed) return []; this.#closed = true; const buf = this.#buffer; @@ -428,7 +439,7 @@ export class HashlineTokenizer { } /** Convenience: feed an entire text and immediately flush. */ - tokenizeAll(text: string): HashlineToken[] { + tokenizeAll(text: string): Token[] { this.reset(); const first = this.feed(text); const last = this.end(); @@ -436,7 +447,7 @@ export class HashlineTokenizer { } /** Stateless one-shot classification. Does not touch the streaming buffer. */ - tokenize(line: string, lineNum = 0): HashlineToken { + tokenize(line: string, lineNum = 0): Token { return classifyLine(line, lineNum); } @@ -456,8 +467,8 @@ export class HashlineTokenizer { ); } - #drainCompleteLines(): HashlineToken[] { - const tokens: HashlineToken[] = []; + #drainCompleteLines(): Token[] { + const tokens: Token[] = []; const buf = this.#buffer; let start = 0; for (let index = 0; index < buf.length; index++) { @@ -471,3 +482,5 @@ export class HashlineTokenizer { return tokens; } } + +export type { ParsedRange } from "./types"; diff --git a/packages/hashline/src/types.ts b/packages/hashline/src/types.ts new file mode 100644 index 000000000..35ddf84db --- /dev/null +++ b/packages/hashline/src/types.ts @@ -0,0 +1,87 @@ +/** + * Pure data types shared across the hashline parser, applier, and patcher. + * Nothing in this file references a filesystem, agent runtime, or schema + * library — keep it that way. + */ + +/** A line-number anchor (1-indexed). */ +export interface Anchor { + line: number; +} + +/** Where an `insert` edit should land relative to existing content. */ +export type Cursor = + | { kind: "bof" } + | { kind: "eof" } + | { kind: "before_anchor"; anchor: Anchor } + | { kind: "after_anchor"; anchor: Anchor }; + +/** + * A single low-level edit produced by the parser and consumed by the applier. + * Multi-line replacements decompose to one `insert` per replacement line plus + * one `delete` per consumed line. + */ +export type Edit = + | { kind: "insert"; cursor: Cursor; text: string; lineNum: number; index: number } + | { kind: "delete"; anchor: Anchor; lineNum: number; index: number; oldAssertion?: string }; + +/** Result of applying a parsed set of edits to a text body. */ +export interface ApplyResult { + /** Post-edit text body. */ + text: string; + /** First line number (1-indexed) that changed, or `undefined` for a no-op apply. */ + firstChangedLine?: number; + /** Diagnostic warnings collected by the applier (auto-absorb, boundary checks, …). */ + warnings?: string[]; +} + +/** Optional knobs forwarded to {@link Edit} application. */ +export interface ApplyOptions { + /** + * When `true`, pure-insert and single-line replacement-boundary duplicates + * are dropped opportunistically. Default `false`: only multi-line block + * duplicates and structural-boundary single lines are absorbed. + */ + autoDropPureInsertDuplicates?: boolean; +} + +/** A parsed `[A..B]` line range. */ +export interface ParsedRange { + start: Anchor; + end: Anchor; +} + +/** Optional hints for {@link splitPatchInput}. */ +export interface SplitOptions { + /** Resolves absolute paths inside hashline headers to cwd-relative form. */ + cwd?: string; + /** + * Fallback path used when the input lacks a `¶PATH` header but contains + * recognizable hashline operations. Lets streaming previews work before + * the model has written the header. + */ + path?: string; +} + +/** Streaming-formatter knobs for {@link streamHashLines}. */ +export interface StreamOptions { + /** First line number to use when formatting (1-indexed, default 1). */ + startLine?: number; + /** Maximum formatted lines per yielded chunk (default 200). */ + maxChunkLines?: number; + /** Maximum UTF-8 bytes per yielded chunk (default 64 KiB). */ + maxChunkBytes?: number; +} + +/** Result of {@link buildCompactDiffPreview}. */ +export interface CompactDiffPreview { + preview: string; + addedLines: number; + removedLines: number; +} + +/** Optional knobs for {@link buildCompactDiffPreview}. Reserved for future use. */ +export interface CompactDiffOptions { + /** Maximum entries kept on each side of an unchanged-context truncation (default 2). */ + maxUnchangedRun?: number; +} diff --git a/packages/hashline/tsconfig.json b/packages/hashline/tsconfig.json new file mode 100644 index 000000000..08130e07c --- /dev/null +++ b/packages/hashline/tsconfig.json @@ -0,0 +1,7 @@ +{ + "extends": "../tsconfig.workspace.json", + "include": [ + "src", + "test" + ] +} diff --git a/packages/hashline/tsconfig.publish.json b/packages/hashline/tsconfig.publish.json new file mode 100644 index 000000000..61f5ee4c6 --- /dev/null +++ b/packages/hashline/tsconfig.publish.json @@ -0,0 +1,22 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "noEmit": false, + "emitDeclarationOnly": true, + "declaration": true, + "declarationMap": false, + "sourceMap": false, + "inlineSources": false, + "rootDir": "src", + "outDir": "dist/types", + "noCheck": true + }, + "include": [ + "src" + ], + "exclude": [ + "dist", + "node_modules", + "test" + ] +} diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index b69cc4bee..54cd068d2 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -1,6 +1,12 @@ # Changelog ## [Unreleased] +### Added + +- Added `Hashline` class with methods to format headers, parse/apply hashline edits, split inputs, compute diffs, generate previews, and recover from stale hashes +- Added `HashlineChunker` class to stream UTF-8 text into numbered hashline chunks incrementally +- Added `HashlineCursorKind`, `HashlineEditKind`, and `HashlineTokenKind` exports for hashline cursor/edit/token discrimination +- Added `unfoldUntilLines` and `unfoldLimitLines` options to `SummaryOptions` to control BFS unfold visibility with an optional hard cap ## [15.5.0] - 2026-05-26 diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 8f5e250f1..132bd28f5 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -1,5 +1,345 @@ /* auto-generated by NAPI-RS */ /* eslint-disable */ +/** + * Hashline patch primitives: parse, apply, hash, diff, recover. + * + * All methods are static (no constructor). Host-agnostic: every input is a + * caller-supplied string or structured value; no filesystem or I/O. + */ +export declare class Hashline { + /** Frozen Lark grammar carried by the Rust crate. */ + static grammar(): string + /** + * Compute the 4-hex section hash for the given text. + * + * Normalizes CRLF and trailing whitespace before hashing so display-trimmed + * lines never invalidate anchors. + */ + static computeFileHash(text: string): string + /** + * Format a `¶path#hash` section header. + * + * Throws when `file_hash` is not exactly four lowercase hex digits. + */ + static formatHeader(path: string, fileHash: string): string + /** Format a single numbered line for read-output display. */ + static formatLine(lineNumber: number, line: string): string + /** Format multi-line text with sequential line-number prefixes. */ + static formatLines(text: string, startLine?: number | undefined | null): string + /** Tokenize hashline input for streaming previews and syntax classification. */ + static tokenize(input: string): Array + /** Parse a hashline diff body into edits and parse-time warnings. */ + static parse(input: string): ParseResult + /** + * Apply pre-parsed edits to text. + * + * Returns the new text plus any apply-time warnings and the first changed + * line number. + */ + static apply(text: string, edits: Array, options?: HashlineApplyOptions | undefined | null): HashlineApplyResult + /** Convenience for parse + apply in one call. */ + static parseAndApply(input: string, text: string, options?: HashlineApplyOptions | undefined | null): HashlineApplyResult + /** + * Split a multi-section hashline input into per-file sections. + * + * Honors `*** Begin Patch` / `*** End Patch` / `*** Abort` envelopes and the + * optional fallback path heuristic. + */ + static split(input: string, options?: SplitHashlineOptions | undefined | null): Array + /** Same as `split`, but returns exactly one section. */ + static splitOne(input: string, options?: SplitHashlineOptions | undefined | null): HashlineInputSection + /** True if the input contains at least one recognizable op line. */ + static containsOps(input: string): boolean + /** + * Run the diff against text. + * + * The host is responsible for reading the file content; everything else + * (hash validation, anchor binding, applier) lives here. + */ + static computeSectionDiff(section: HashlineInputSection, text: string, options?: HashlineApplyOptions | undefined | null): DiffResult + /** + * Compute a full unified diff for a `(input, text)` pair. + * + * Errors when the input expands to anything other than exactly one section. + */ + static computeDiff(input: string, fallbackPath: string | undefined | null, text: string, options?: HashlineApplyOptions | undefined | null): DiffResult + /** Generate the compact `+N:` / `-N:` / ` N:` preview from a unified diff. */ + static compactPreview(diff: string): CompactHashlineDiffPreview + /** Strip leading `LINE:` / `+LINE:` prefixes from echoed read-output text. */ + static stripPrefixes(lines: Array): Array + /** + * Like `strip_prefixes`, but only when every non-blank line carries the + * hashline prefix. + */ + static stripHashlinePrefixes(lines: Array): Array + /** + * Normalize line payloads by stripping echoed prefixes. + * + * `null` / `undefined` yield an empty array; a multiline string is split on + * ` + `. + */ + static parseText(text?: string | undefined | null): Array + /** + * Attempt three-way merge recovery against cached snapshots. + * + * Returns `null` when no recovery is possible. + */ + static recover(args: HashlineRecoveryArgs): HashlineRecoveryResult | null + /** + * One-shot streaming chunker. + * + * For incremental streaming, construct `HashlineChunker` directly. + */ + static streamChunks(chunks: Array, options?: HashlineStreamOptions | undefined | null): Array +} + +/** Stateful chunker that formats a UTF-8 stream into numbered hashline chunks. */ +export declare class HashlineChunker { + /** Open a new chunker with the given streaming options. */ + constructor(options?: HashlineStreamOptions | undefined | null) + /** Push UTF-8 text and return any flushed chunks. */ + push(chunk: string): Array + /** Finish the stream and drain remaining buffered output. */ + finish(): Array +} + +/** One line anchor used by hashline cursors and ranges. */ +export interface Anchor { + /** 1-indexed file line number. */ + line: number +} + +/** Compact preview summary derived from a unified diff. */ +export interface CompactHashlineDiffPreview { + /** Collapsed preview body. */ + preview: string + /** Count of added lines in the underlying diff. */ + addedLines: number + /** Count of removed lines in the underlying diff. */ + removedLines: number +} + +/** Unified diff output. */ +export interface DiffResult { + /** Full unified diff. */ + diff: string + /** First changed line in the new file, when known. */ + firstChangedLine?: number +} + +/** Cached file snapshot used for stale-hash recovery. */ +export interface FileReadSnapshot { + /** Sparse or contiguous line samples from the cached read. */ + lines: Array + /** Optional full file text when the read captured it. */ + fullText?: string + /** Optional cached file hash for the snapshot text. */ + fileHash?: string +} + +/** Options for applying edits. */ +export interface HashlineApplyOptions { + /** Enable duplicate-boundary absorption for pure inserts. */ + autoDropPureInsertDuplicates?: boolean +} + +/** Apply output from `Hashline.apply` / `Hashline.parseAndApply`. */ +export interface HashlineApplyResult { + /** Full post-apply file text. */ + lines: string + /** First changed line in the resulting file, when known. */ + firstChangedLine?: number + /** Apply-time warnings. */ + warnings?: Array + /** Edits that matched but changed nothing. */ + noopEdits?: Array +} + +/** Tagged cursor location for insert operations. */ +export interface HashlineCursor { + /** Cursor discriminator. */ + kind: HashlineCursorKind + /** Anchor payload for `before_anchor` / `after_anchor` cursors. */ + anchor?: Anchor +} + +/** Discriminator for a hashline cursor. */ +export declare enum HashlineCursorKind { + /** Beginning of file. */ + Bof = 'bof', + /** End of file. */ + Eof = 'eof', + /** Insert before the given anchor. */ + BeforeAnchor = 'before_anchor', + /** Insert after the given anchor. */ + AfterAnchor = 'after_anchor' +} + +/** Parsed hashline edit. */ +export interface HashlineEdit { + /** Edit discriminator. */ + kind: HashlineEditKind + /** Source line number in the hashline input. */ + lineNum: number + /** Stable edit index within the parsed patch. */ + index: number + /** Insert cursor when `kind === "insert"`. */ + cursor?: HashlineCursor + /** Insert payload when `kind === "insert"`. */ + text?: string + /** Delete anchor when `kind === "delete"`. */ + anchor?: Anchor + /** Optional current-file assertion captured from delete payload. */ + oldAssertion?: string +} + +/** Discriminator for a parsed edit. */ +export declare enum HashlineEditKind { + /** Insert text at a cursor. */ + Insert = 'insert', + /** Delete one anchored line. */ + Delete = 'delete' +} + +/** One `¶path#hash` section. */ +export interface HashlineInputSection { + /** Section path. */ + path: string + /** Optional 4-hex file hash. */ + fileHash?: string + /** Raw diff body for this section. */ + diff: string +} + +/** One no-op edit explanation from apply-time diagnostics. */ +export interface HashlineNoopEdit { + /** Index of the no-op edit in the parsed edit list. */ + editIndex: number + /** Human-readable location label. */ + loc: string + /** Why the edit was a no-op. */ + reason: string + /** Current content that caused the no-op. */ + current: string +} + +/** Inclusive line range used by replace/delete operations and tokenizer output. */ +export interface HashlineRange { + /** First line in the inclusive range. */ + start: Anchor + /** Last line in the inclusive range. */ + end: Anchor +} + +/** Inputs for stale-hash recovery. */ +export interface HashlineRecoveryArgs { + /** Path of the file being recovered. */ + path: string + /** Current on-disk text. */ + currentText: string + /** Section hash the edits were anchored against. */ + fileHash: string + /** Parsed edits from the stale section. */ + edits: Array + /** Optional earlier head snapshot from the same session chain. */ + headSnapshot?: FileReadSnapshot + /** Required target snapshot for the stale file hash. */ + targetSnapshot?: FileReadSnapshot + /** Apply-time options. */ + options?: HashlineApplyOptions +} + +/** Successful stale-hash recovery result. */ +export interface HashlineRecoveryResult { + /** Full recovered file text. */ + lines: string + /** First changed line in the recovered file, when known. */ + firstChangedLine?: number + /** Recovery warnings explaining the merge path taken. */ + warnings: Array +} + +/** Options for UTF-8 stream chunking. */ +export interface HashlineStreamOptions { + /** First numbered output line. */ + startLine?: number + /** Maximum lines per flushed chunk. */ + maxChunkLines?: number + /** Maximum bytes per flushed chunk. */ + maxChunkBytes?: number +} + +/** One tokenizer token. */ +export interface HashlineToken { + /** Token discriminator. */ + kind: HashlineTokenKind + /** 1-indexed source line number. */ + lineNum: number + /** Header path when `kind === "header"`. */ + path?: string + /** Header hash when `kind === "header"`. */ + fileHash?: string + /** Insert cursor when `kind === "op-insert"`. */ + cursor?: HashlineCursor + /** Replace/delete range when applicable. */ + range?: HashlineRange + /** Inline payload captured on the op line. */ + inlineBody?: string + /** Whether a delete op was followed by payload-looking text. */ + trailingPayload?: boolean + /** Payload or raw text. */ + text?: string +} + +/** Discriminator for tokenizer output. */ +export declare enum HashlineTokenKind { + /** Empty line. */ + Blank = 'blank', + /** `*** Begin Patch`. */ + EnvelopeBegin = 'envelope-begin', + /** `*** End Patch`. */ + EnvelopeEnd = 'envelope-end', + /** `*** Abort`. */ + Abort = 'abort', + /** `¶path#hash` header. */ + Header = 'header', + /** Insert op (`↑` / `↓`). */ + OpInsert = 'op-insert', + /** Replace op (`:`). */ + OpReplace = 'op-replace', + /** Delete op (`!`). */ + OpDelete = 'op-delete', + /** Payload continuation line. */ + Payload = 'payload', + /** Unclassified raw input line. */ + Raw = 'raw' +} + +/** Parse output from `Hashline.parse`. */ +export interface ParseResult { + /** Parsed edits in execution order. */ + edits: Array + /** Parse-time warnings. */ + warnings: Array +} + +/** One 1-indexed line entry in a cached file snapshot. */ +export interface SnapshotLine { + /** 1-indexed line number. */ + line: number + /** Exact line text, without the trailing ` + `. */ + text: string +} + +/** Options for splitting multi-section input. */ +export interface SplitHashlineOptions { + /** Working directory used to relativize absolute section paths. */ + cwd?: string + /** Fallback path when the input omits an explicit `¶path` header. */ + path?: string +} /** * Long-lived macOS appearance observer. * @@ -1296,6 +1636,16 @@ export interface SummaryOptions { minBodyLines?: number /** Minimum total comment lines before eliding a multiline block comment. */ minCommentLines?: number + /** + * Target visible-line count for BFS unfold. `None` or `0` keeps only + * the outermost elisions (no progressive unfolding). + */ + unfoldUntilLines?: number + /** + * Hard ceiling for BFS unfold. Defaults to `unfold_until_lines * 2` + * when omitted. + */ + unfoldLimitLines?: number } export interface SummaryResult { diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 23bfd40d3..3eb8f3de0 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -16,6 +16,8 @@ import { loadNative } from "./loader-state.js"; const nativeBindings = loadNative(); // --- generated native exports (do not edit) --- // classes +export const Hashline = nativeBindings.Hashline; +export const HashlineChunker = nativeBindings.HashlineChunker; export const MacAppearanceObserver = nativeBindings.MacAppearanceObserver; export const MacOSPowerAssertion = nativeBindings.MacOSPowerAssertion; export const Process = nativeBindings.Process; @@ -65,6 +67,28 @@ export const visibleWidth = nativeBindings.visibleWidth; export const wrapTextWithAnsi = nativeBindings.wrapTextWithAnsi; // string/numeric enums (napi-rs string_enum produces TS-only const enum) +export const HashlineCursorKind = { + Bof: "bof", + Eof: "eof", + BeforeAnchor: "before_anchor", + AfterAnchor: "after_anchor", +}; +export const HashlineEditKind = { + Insert: "insert", + Delete: "delete", +}; +export const HashlineTokenKind = { + Blank: "blank", + EnvelopeBegin: "envelope-begin", + EnvelopeEnd: "envelope-end", + Abort: "abort", + Header: "header", + OpInsert: "op-insert", + OpReplace: "op-replace", + OpDelete: "op-delete", + Payload: "payload", + Raw: "raw", +}; export const AstMatchStrictness = { Cst: "cst", Smart: "smart", diff --git a/scripts/ci-release-publish.ts b/scripts/ci-release-publish.ts index 2995f7222..51291ed70 100644 --- a/scripts/ci-release-publish.ts +++ b/scripts/ci-release-publish.ts @@ -46,6 +46,7 @@ const packages: PublishPackage[] = [ { dir: "packages/ai", kind: "typescript" }, { dir: "packages/natives", kind: "native" }, { dir: "packages/tui", kind: "typescript" }, + { dir: "packages/hashline", kind: "typescript" }, { dir: "packages/stats", kind: "typescript",