From 01c34db450cc5fdf2bcb51ade5a4f02d08877b80 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 29 May 2026 17:57:03 +0200 Subject: [PATCH] feat(hashline): added full-file hash snapshots with 4-hex tags - Replaced snapshot internals with full-file records and removed contiguous/sparse snapshot APIs. - Added file-hash normalization, computed `computeFileHash`, and updated grammar/messages to 4-hex tags. - Simplified recovery by checking whole-file hashes first, then applying merge-replay fallback after mismatches. - Updated coding-agent tools to use `record`/`recordFileSnapshot` and skip hash headers for unsnapshotted large files. - Expanded patcher and snapshot tests to verify 4-hex anchors, hash deduplication, and cache-capped behavior. --- bun.lock | 1 + packages/coding-agent/CHANGELOG.md | 7 +- .../src/edit/file-snapshot-store.ts | 34 ++ .../coding-agent/src/edit/hashline/diff.ts | 11 +- packages/coding-agent/src/edit/renderer.ts | 2 +- packages/coding-agent/src/tools/ast-edit.ts | 2 +- packages/coding-agent/src/tools/ast-grep.ts | 23 +- packages/coding-agent/src/tools/read.ts | 56 +- packages/coding-agent/src/tools/search.ts | 33 +- packages/coding-agent/src/tools/write.ts | 4 +- .../coding-agent/src/utils/file-mentions.ts | 4 +- .../coding-agent/test/core/hashline.test.ts | 70 +-- packages/coding-agent/test/edit-diff.test.ts | 2 +- .../read-column-truncation-snapshot.test.ts | 12 +- .../coding-agent/test/tools/ast-edit.test.ts | 4 +- .../coding-agent/test/tools/ast-grep.test.ts | 4 +- .../test/tools/search-internal-urls.test.ts | 6 +- .../test/tools/search-path-lists.test.ts | 30 +- .../test/write-hashline-header.test.ts | 4 +- packages/hashline/CHANGELOG.md | 15 +- .../hashline/bench/recovery-session-chain.ts | 4 +- packages/hashline/package.json | 3 +- packages/hashline/src/format.ts | 33 +- packages/hashline/src/grammar.lark | 2 +- packages/hashline/src/input.ts | 6 +- packages/hashline/src/mismatch.ts | 7 +- packages/hashline/src/patcher.ts | 60 +-- packages/hashline/src/recovery.ts | 85 +-- packages/hashline/src/snapshots.ts | 482 ++++-------------- .../hashline/test/boundary-repair.test.ts | 2 +- packages/hashline/test/patcher.test.ts | 73 +-- .../test/recovery-session-chain.test.ts | 4 +- packages/hashline/test/snapshots.test.ts | 145 +++--- scripts/session-stats/sync.py | 114 ++++- 34 files changed, 517 insertions(+), 827 deletions(-) diff --git a/bun.lock b/bun.lock index 590517b7b..f641eada3 100644 --- a/bun.lock +++ b/bun.lock @@ -84,6 +84,7 @@ "version": "15.5.12", "dependencies": { "diff": "catalog:", + "lru-cache": "catalog:", }, "devDependencies": { "@types/bun": "catalog:", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b76af5ebf..e2b4956de 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,11 +1,16 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Changed hashline edit syntax to verb-based v4: body-bearing ops are `replace N..M:`, `insert before N:`, `insert after N:`, `insert head:`, and `insert tail:`, while bodyless `delete N..M` handles deletion. Removed `>A..B` repeat rows and the old `prepend:` / `append:` virtual insert headers; `-` rows remain rejected with a teaching error. +### Changed + +- Changed hashline tag generation to use full-file snapshots for read/search/ast-grep and related outputs, so hashline anchors now validate only when the complete file matches +- Changed hashline tagging to omit file headers for files over 4 MiB or that cannot be snapshotted, so those files are returned without editable hashline anchors +- Changed hashline context generation for line edits from partial/sparse snippets to complete-file fingerprints, reducing stale anchors for partially read files + ### Fixed - Restored automatic repair of `edit` range hunks that break bracket balance — the failure class that previously left a duplicated closing line (a `` / `);` / `}` echoed just below the range) or dropped one (the range swallowed a `});` the payload never restated), leaving the file syntactically broken until a follow-up edit. The hashline applier now normalizes each replacement so its payload preserves the deleted region's delimiter balance, dropping a duplicated bordering closer or sparing a deleted one, and surfaces a warning on the tool result. Always on and balance-validated (no `edit.hashlineAutoDropPureInsertDuplicates` setting); see `@oh-my-pi/hashline` for the contract. diff --git a/packages/coding-agent/src/edit/file-snapshot-store.ts b/packages/coding-agent/src/edit/file-snapshot-store.ts index ef0d5ae42..ca51dbd11 100644 --- a/packages/coding-agent/src/edit/file-snapshot-store.ts +++ b/packages/coding-agent/src/edit/file-snapshot-store.ts @@ -9,6 +9,15 @@ * is wiring it onto the per-session owner object. */ import { InMemorySnapshotStore } from "@oh-my-pi/hashline"; +import { normalizeToLF } from "./normalize"; + +/** + * Upper bound on the file size we snapshot. A section tag is a content hash of + * the *whole* file, so minting one means holding the full normalized text in + * the store. Files above this cap emit no `¶path#tag` header — line-anchored + * editing of multi-megabyte files is out of scope under the full-content model. + */ +export const SNAPSHOT_MAX_BYTES = 4 * 1024 * 1024; interface FileSnapshotStoreOwner { fileSnapshotStore?: InMemorySnapshotStore; @@ -23,3 +32,28 @@ export function getFileSnapshotStore(session: FileSnapshotStoreOwner): InMemoryS if (!session.fileSnapshotStore) session.fileSnapshotStore = new InMemorySnapshotStore(); return session.fileSnapshotStore; } + +/** + * Read the full text of `absolutePath` (within {@link SNAPSHOT_MAX_BYTES}), + * record it as a version snapshot, and return its content-hash tag. Returns + * `undefined` when the file exceeds the cap or cannot be read — callers then + * omit the section header so the model never sees a tag it can't anchor against. + * + * Producers that only displayed a slice of the file (range reads, search hits) + * use this to mint a whole-file tag: the displayed lines stay partial, but the + * tag fingerprints the entire file so a follow-up edit anchored at any line + * validates whenever the live file is byte-identical to what was read. + */ +export async function recordFileSnapshot( + session: FileSnapshotStoreOwner, + absolutePath: string, +): Promise { + try { + const file = Bun.file(absolutePath); + if (file.size > SNAPSHOT_MAX_BYTES) return undefined; + const normalized = normalizeToLF(await file.text()); + return getFileSnapshotStore(session).record(absolutePath, normalized); + } catch { + return undefined; + } +} diff --git a/packages/coding-agent/src/edit/hashline/diff.ts b/packages/coding-agent/src/edit/hashline/diff.ts index 5c7359035..38f3cbf68 100644 --- a/packages/coding-agent/src/edit/hashline/diff.ts +++ b/packages/coding-agent/src/edit/hashline/diff.ts @@ -44,14 +44,9 @@ function hasAnchorScoped(section: PatchSection): boolean { return section.hasAnchorScopedEdit; } -function snapshotMatchesCurrent(snapshot: Snapshot, currentText: string, anchorLines: readonly number[]): boolean { - if (snapshot.fullText !== undefined) return snapshot.fullText === currentText; - for (const lineNumber of anchorLines) { - if (snapshot.get(lineNumber) === undefined) return false; - } - return snapshot.matchesLiveFile(currentText.split("\n")); +function snapshotMatchesCurrent(snapshot: Snapshot, currentText: string): boolean { + return snapshot.text === currentText; } - function validateSectionHash( section: PatchSection, absolutePath: string, @@ -64,7 +59,7 @@ function validateSectionHash( : null; } const snapshot = snapshots.byHash(absolutePath, section.fileHash); - if (snapshot && snapshotMatchesCurrent(snapshot, text, section.collectAnchorLines())) return null; + if (snapshot && snapshotMatchesCurrent(snapshot, text)) return null; return `Hashline snapshot tag mismatch for ${section.path}: section is bound to #${section.fileHash}, but current file does not match that snapshot; re-read and try again.`; } diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 1e415a671..ec1c1bac4 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -312,7 +312,7 @@ const MISSING_APPLY_PATCH_END_ERROR = "The last line of the patch must be '*** E function normalizeHashlineInputPreviewPath(rawPath: string): string { const trimmed = rawPath.trim(); - const hashStart = /#[0-9a-fA-F]{3}$/u.exec(trimmed)?.index; + const hashStart = /#[0-9a-fA-F]{4}$/u.exec(trimmed)?.index; const withoutHash = hashStart === undefined ? trimmed : trimmed.slice(0, hashStart); if (withoutHash.length < 2) return withoutHash; const first = withoutHash[0]; diff --git a/packages/coding-agent/src/tools/ast-edit.ts b/packages/coding-agent/src/tools/ast-edit.ts index 60c00a6c9..08c679aa4 100644 --- a/packages/coding-agent/src/tools/ast-edit.ts +++ b/packages/coding-agent/src/tools/ast-edit.ts @@ -290,7 +290,7 @@ export class AstEditTool implements AgentTool(); - const snapshotStore = useHashLines ? getFileSnapshotStore(this.session) : undefined; + const hashContexts = new Map(); if (useHashLines) { for (const relativePath of fileList) { const absolutePath = path.resolve(this.session.cwd, relativePath); - try { - await access(absolutePath, constants.R_OK); - hashContexts.set(relativePath, { absolutePath }); - } catch { - // Best-effort: if a file disappears between ast-grep and rendering, emit plain line output. - } + // Whole-file content tag: any anchor validates while the file is + // unchanged; over-cap / unreadable files get no tag (plain output). + const tag = await recordFileSnapshot(this.session, absolutePath); + if (tag) hashContexts.set(relativePath, { tag }); } } const outputLines: string[] = []; @@ -246,7 +241,6 @@ export class AstGrepTool implements AgentTool = []; for (const match of fileMatches) { const matchLines = match.text.split("\n"); for (let index = 0; index < matchLines.length; index++) { @@ -257,7 +251,6 @@ export class AstGrepTool implements AgentTool 0) { const serializedMeta = Object.entries(match.metaVariables) @@ -269,10 +262,6 @@ export class AstGrepTool implements AgentTool 0) { - const tag = snapshotStore?.recordSparse(hashContext.absolutePath, cacheEntries); - if (tag) hashContext.tag = tag; - } return { model: modelOut, display: displayOut }; }; diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 4b003a385..20efd4e98 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -9,7 +9,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { getRemoteDir, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { getFileSnapshotStore } from "../edit/file-snapshot-store"; +import { getFileSnapshotStore, recordFileSnapshot } from "../edit/file-snapshot-store"; import { normalizeToLF } from "../edit/normalize"; import { isNotebookPath, readEditableNotebookText } from "../edit/notebook"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; @@ -130,9 +130,7 @@ function recordFullHashlineContext( ): HashlineHeaderContext | undefined { if (!absolutePath || !path.isAbsolute(absolutePath)) return undefined; const normalized = normalizeToLF(fullText); - const tag = getFileSnapshotStore(session).recordContiguous(absolutePath, 1, normalized.split("\n"), { - fullText: normalized, - }); + const tag = getFileSnapshotStore(session).record(absolutePath, normalized); return { header: formatHashlineHeader(displayPath, tag), tag, @@ -1033,7 +1031,6 @@ export class ReadTool implements AgentTool { const shouldAddHashLines = !rawSelector && displayMode.hashLines; const shouldAddLineNumbers = rawSelector ? false : shouldAddHashLines ? false : displayMode.lineNumbers; - const sparseSnapshotEntries: Array = []; const maxColumns = resolveOutputMaxColumns(this.session.settings); const blocks: string[] = []; @@ -1063,10 +1060,8 @@ export class ReadTool implements AgentTool { } const collectedLines = streamResult.lines; - // Column truncation is display-only. The snapshot (sparseSnapshotEntries) - // MUST hold on-disk content so later edits can verify line content against - // the live file. Stamping ellipsis-truncated lines into the snapshot makes - // every long-line file uneditable on the next edit attempt. + // Column truncation is display-only; clone before stamping ellipsis so + // the original on-disk lines stay intact for display reconstruction. let displayLines: string[] = collectedLines; if (!rawSelector && maxColumns > 0) { let cloned: string[] | undefined; @@ -1080,19 +1075,16 @@ export class ReadTool implements AgentTool { } if (cloned) displayLines = cloned; } - - for (let index = 0; index < collectedLines.length; index++) { - sparseSnapshotEntries.push([range.startLine + index, collectedLines[index]]); - } - const blockText = displayLines.join("\n"); blocks.push(formatTextWithMode(blockText, range.startLine, shouldAddHashLines, shouldAddLineNumbers)); } let outputText = blocks.join("\n\n…\n\n"); - if (shouldAddHashLines && sparseSnapshotEntries.length > 0 && outputText) { - const tag = getFileSnapshotStore(this.session).recordSparse(absolutePath, sparseSnapshotEntries); - outputText = `${formatHashlineHeader(formatPathRelativeToCwd(absolutePath, this.session.cwd), tag)}\n${outputText}`; + if (shouldAddHashLines && outputText) { + const tag = await recordFileSnapshot(this.session, absolutePath); + if (tag) { + outputText = `${formatHashlineHeader(formatPathRelativeToCwd(absolutePath, this.session.cwd), tag)}\n${outputText}`; + } } if (notices.length > 0) { outputText = outputText ? `${outputText}\n${notices.join("\n")}` : notices.join("\n"); @@ -1905,17 +1897,17 @@ export class ReadTool implements AgentTool { const shouldAddLineNumbers = rawSelector ? false : shouldAddHashLines ? false : displayMode.lineNumbers; let hashContext: HashlineHeaderContext | undefined; if (shouldAddHashLines && collectedLines.length > 0 && !firstLineExceedsLimit) { - const store = getFileSnapshotStore(this.session); - const tag = - offset === undefined && limit === undefined && !wasTruncated - ? (() => { - const normalized = normalizeToLF(collectedLines.join("\n")); - return store.recordContiguous(absolutePath, 1, normalized.split("\n"), { - fullText: normalized, - }); - })() - : store.recordContiguous(absolutePath, startLineDisplay, collectedLines); - hashContext = hashlineHeaderContext(formatPathRelativeToCwd(absolutePath, this.session.cwd), tag); + // The tag is a content hash of the WHOLE file. A whole-file read + // already holds every line in memory; a range read re-reads the + // file (bounded by SNAPSHOT_MAX_BYTES) so the tag fingerprints the + // full file and any anchor validates while the file is unchanged. + const isWholeFile = offset === undefined && limit === undefined && !wasTruncated; + const tag = isWholeFile + ? getFileSnapshotStore(this.session).record(absolutePath, normalizeToLF(collectedLines.join("\n"))) + : await recordFileSnapshot(this.session, absolutePath); + if (tag) { + hashContext = hashlineHeaderContext(formatPathRelativeToCwd(absolutePath, this.session.cwd), tag); + } } let capturedDisplayContent: { text: string; startLine: number } | undefined; @@ -2060,11 +2052,9 @@ export class ReadTool implements AgentTool { const shouldAddLineNumbers = shouldAddHashLines ? false : displayMode.lineNumbers; const rawText = region.lines.join("\n"); - const hashContext = shouldAddHashLines - ? hashlineHeaderContext( - formatPathRelativeToCwd(entry.absolutePath, this.session.cwd), - getFileSnapshotStore(this.session).recordContiguous(entry.absolutePath, region.startLine, region.lines), - ) + const tag = shouldAddHashLines ? await recordFileSnapshot(this.session, entry.absolutePath) : undefined; + const hashContext = tag + ? hashlineHeaderContext(formatPathRelativeToCwd(entry.absolutePath, this.session.cwd), tag) : undefined; const formattedBody = formatTextWithMode(rawText, region.startLine, shouldAddHashLines, shouldAddLineNumbers); const formattedText = prependHashlineHeader(formattedBody, hashContext); diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index c1fc405d5..679447734 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -1,5 +1,4 @@ -import { constants } from "node:fs"; -import { access, mkdtemp, rm, stat, writeFile } from "node:fs/promises"; +import { mkdtemp, rm, stat, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import * as path from "node:path"; import { formatHashlineHeader } from "@oh-my-pi/hashline"; @@ -9,7 +8,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { getFileSnapshotStore } from "../edit/file-snapshot-store"; +import { recordFileSnapshot } from "../edit/file-snapshot-store"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import type { Theme } from "../modes/theme/theme"; import searchDescription from "../prompts/tools/search.md" with { type: "text" }; @@ -610,19 +609,17 @@ export class SearchTool implements AgentTool(); - const snapshotStore = baseDisplayMode.hashLines ? getFileSnapshotStore(this.session) : undefined; + const hashContexts = new Map(); if (baseDisplayMode.hashLines) { for (const relativePath of fileList) { if (archiveDisplaySet.has(relativePath)) continue; const absoluteFilePath = path.resolve(this.session.cwd, relativePath); if (immutableSourcePaths.has(absoluteFilePath)) continue; - try { - await access(absoluteFilePath, constants.R_OK); - hashContexts.set(relativePath, { absolutePath: absoluteFilePath }); - } catch { - // Best-effort: if the file disappeared between grep and render, fall back to plain line output. - } + // Mint a whole-file content tag so any anchor validates while the + // file is unchanged; over-cap / unreadable files get no tag (and + // therefore plain, non-editable line output). + const tag = await recordFileSnapshot(this.session, absoluteFilePath); + if (tag) hashContexts.set(relativePath, { tag }); } } const renderMatchesForFile = (relativePath: string): { model: string[]; display: string[] } => { @@ -641,40 +638,34 @@ export class SearchTool implements AgentTool = []; let lastEmittedLine: number | undefined; const gutterPad = " ".repeat(lineNumberWidth + 1); for (const match of fileMatches) { - const pushLine = (lineNumber: number, line: string, isMatch: boolean, recordable: boolean) => { + const pushLine = (lineNumber: number, line: string, isMatch: boolean) => { if (lastEmittedLine !== undefined && lineNumber > lastEmittedLine + 1) { modelOut.push("..."); displayOut.push(`${gutterPad}│...`); } modelOut.push(formatMatchLine(lineNumber, line, isMatch, { useHashLines })); displayOut.push(formatCodeFrameLine(isMatch ? "*" : " ", lineNumber, line, lineNumberWidth)); - if (recordable) cacheEntries.push([lineNumber, line] as const); lastEmittedLine = lineNumber; }; if (match.contextBefore) { for (const ctx of match.contextBefore) { - pushLine(ctx.lineNumber, ctx.line, false, true); + pushLine(ctx.lineNumber, ctx.line, false); } } - pushLine(match.lineNumber, match.line, true, !match.truncated); + pushLine(match.lineNumber, match.line, true); if (match.truncated) { linesTruncated = true; } if (match.contextAfter) { for (const ctx of match.contextAfter) { - pushLine(ctx.lineNumber, ctx.line, false, true); + pushLine(ctx.lineNumber, ctx.line, false); } } fileMatchCounts.set(relativePath, (fileMatchCounts.get(relativePath) ?? 0) + 1); } - if (cacheEntries.length > 0 && hashContext) { - const tag = snapshotStore?.recordSparse(hashContext.absolutePath, cacheEntries); - if (tag) hashContext.tag = tag; - } return { model: modelOut, display: displayOut }; }; if (isDirectory) { diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 0ef9e588e..9c5bb95fb 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -130,9 +130,7 @@ function stripWriteContent(session: ToolSession, content: string): { text: strin function maybeWriteSnapshotHeader(session: ToolSession, absolutePath: string, content: string): string | undefined { if (!resolveFileDisplayMode(session).hashLines) return undefined; const normalized = normalizeToLF(content); - const tag = getFileSnapshotStore(session).recordContiguous(absolutePath, 1, normalized.split("\n"), { - fullText: normalized, - }); + const tag = getFileSnapshotStore(session).record(absolutePath, normalized); return formatHashlineHeader(formatPathRelativeToCwd(absolutePath, session.cwd), tag); } diff --git a/packages/coding-agent/src/utils/file-mentions.ts b/packages/coding-agent/src/utils/file-mentions.ts index 6e156e032..517b3d394 100644 --- a/packages/coding-agent/src/utils/file-mentions.ts +++ b/packages/coding-agent/src/utils/file-mentions.ts @@ -359,9 +359,7 @@ export async function generateFileMentionMessages( const normalized = snapshotStore ? normalizeToLF(content) : content; let { output, lineCount } = buildTextOutput(normalized); if (snapshotStore) { - const tag = snapshotStore.recordContiguous(absolutePath, 1, normalized.split("\n"), { - fullText: normalized, - }); + const tag = snapshotStore.record(absolutePath, normalized); output = `${formatHashlineHeader(resolvedPath, tag)}\n${formatNumberedLines(output)}`; } files.push({ path: resolvedPath, content: output, lineCount }); diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index 71c703cf4..cbaef2609 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -96,7 +96,7 @@ function tag(line: number, _content: string): string { } function recordFullSnapshot(cache: FileReadCache, filePath: string, fullText: string): string { - return cache.recordContiguous(filePath, 1, fullText.split("\n"), { fullText }); + return cache.record(filePath, fullText); } function header(filePath: string, tag: string): string { @@ -546,10 +546,10 @@ describe("hashline — snapshot tag binding", () => { describe("splitHashlineInput — @ headers", () => { it("extracts path, snapshot tag, and diff body from @path#tag header", () => { - const input = [`¶src/foo.ts#0A3`, `${sameLineRange(tag(2, "bbb"))}`, repl("BBB")].join("\n"); + const input = [`¶src/foo.ts#1A2B`, `${sameLineRange(tag(2, "bbb"))}`, repl("BBB")].join("\n"); expect(splitHashlineInput(input)).toEqual({ path: "src/foo.ts", - fileHash: "0A3", + fileHash: "1A2B", diff: `${sameLineRange(tag(2, "bbb"))}\n${repl("BBB")}`, }); }); @@ -679,7 +679,7 @@ describe("hashline executor", () => { await Bun.write(bPath, "bbb\n"); const session = makeHashlineSession(tempDir); const aTag = recordFullSnapshot(getFileReadCache(session), aPath, "aaa\n"); - const bHeader = "¶b.ts#fff"; + const bHeader = "¶b.ts#FFFF"; const input = [ header("a.ts", aTag), `${sameLineRange(tag(1, "aaa"))}`, @@ -890,9 +890,10 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { await Bun.write(filePath, v0Text); const session = makeHashlineSession(tempDir); - // Cache only covers the first three lines — enough to retain the snapshot tag - // but not enough to synthesize the requested pre-edit snapshot. - const v0Tag = getFileReadCache(session).recordContiguous(filePath, 1, v0Lines.slice(0, 3)); + // Record the full V0 snapshot. The external change below rewrites the + // exact line the model anchors against, so neither the 3-way merge nor + // session replay can land — recovery must decline. + const v0Tag = recordFullSnapshot(getFileReadCache(session), filePath, v0Text); const v1Lines = [...v0Lines]; v1Lines[5] = "L6-CHANGED"; @@ -932,7 +933,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { const a = new FileReadCache(); const b = new FileReadCache(); const fakePath = "/tmp/__hashline-cache-isolation__.ts"; - a.recordContiguous(fakePath, 1, ["x", "y", "z"]); + a.record(fakePath, "x\ny\nz\n"); expect(a.head(fakePath)).not.toBeNull(); expect(b.head(fakePath)).toBeNull(); }); @@ -957,9 +958,7 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(await Bun.file(filePath).text()).toBe(v1Text); const v1Tag = recordFullSnapshot(getFileReadCache(session), filePath, v1Text); const snap = getFileReadCache(session).head(filePath); - expect(snap?.get(1)).toBe("alpha"); - expect(snap?.get(2)).toBe("BETA"); - expect(snap?.get(3)).toBe("gamma"); + expect(snap?.text).toBe(v1Text); // External actor insert heads 7 lines after the edit. Anchors authored // against V1 (the post-edit state the model just observed) no longer @@ -1033,47 +1032,26 @@ describe("hashline — anchor-stale recovery via read snapshot cache", () => { expect(recovered?.lines).toContain("L10-EDITED"); }); - it("retains older snapshot tags in the per-path snapshot ring", () => { + it("retains older versions per path so stale tags still resolve", () => { const cache = new FileReadCache(); - const fakePath = "/tmp/__hashline-cache-ring__.ts"; + const fakePath = "/tmp/__hashline-cache-history__.ts"; const oneTag = recordFullSnapshot(cache, fakePath, "one\n"); const twoTag = recordFullSnapshot(cache, fakePath, "two\n"); recordFullSnapshot(cache, fakePath, "three\n"); - expect(cache.head(fakePath)?.fullText).toBe("three\n"); - expect(cache.byHash(fakePath, oneTag)?.fullText).toBe("one\n"); - expect(cache.byHash(fakePath, twoTag)?.fullText).toBe("two\n"); + expect(cache.head(fakePath)?.text).toBe("three\n"); + expect(cache.byHash(fakePath, oneTag)?.text).toBe("one\n"); + expect(cache.byHash(fakePath, twoTag)?.text).toBe("two\n"); }); - - it("pushes a fresh snapshot when newly recorded lines disagree on overlap", () => { - const cache = new FileReadCache(); - const fakePath = "/tmp/__hashline-cache-conflict__.ts"; - cache.recordContiguous(fakePath, 1, ["a", "b", "c", "d", "e"]); - cache.recordSparse(fakePath, [ - [3, "c"], - [4, "D-CHANGED"], - [5, "e"], - [6, "f"], - [7, "g"], - ]); - - const snap = cache.head(fakePath); - expect(snap).not.toBeNull(); - // Old entries dropped; only the divergent record's entries remain. - expect(snap?.get(1)).toBeUndefined(); - expect(snap?.get(2)).toBeUndefined(); - expect(snap?.get(4)).toBe("D-CHANGED"); - expect(snap?.get(7)).toBe("g"); - }); - - it("keeps independently tracked path rings without an LRU cap", () => { - const cache = new FileReadCache(); - for (let i = 0; i < 32; i++) { - cache.recordContiguous(`/tmp/file-${i}.ts`, 1, [`x${i}`]); + it("evicts the least-recently-used path beyond the LRU cap", () => { + const cache = new FileReadCache({ maxPaths: 4 }); + for (let i = 0; i < 6; i++) { + recordFullSnapshot(cache, `/tmp/file-${i}.ts`, `x${i}\n`); } - expect(cache.head("/tmp/file-0.ts")?.get(1)).toBe("x0"); - expect(cache.head("/tmp/file-1.ts")?.get(1)).toBe("x1"); - expect(cache.head("/tmp/file-2.ts")?.get(1)).toBe("x2"); - expect(cache.head("/tmp/file-31.ts")?.get(1)).toBe("x31"); + // The two oldest paths aged out; the four most-recent survive. + expect(cache.head("/tmp/file-0.ts")).toBeNull(); + expect(cache.head("/tmp/file-1.ts")).toBeNull(); + expect(cache.head("/tmp/file-2.ts")?.text).toBe("x2\n"); + expect(cache.head("/tmp/file-5.ts")?.text).toBe("x5\n"); }); }); diff --git a/packages/coding-agent/test/edit-diff.test.ts b/packages/coding-agent/test/edit-diff.test.ts index 4856e07ff..634902777 100644 --- a/packages/coding-agent/test/edit-diff.test.ts +++ b/packages/coding-agent/test/edit-diff.test.ts @@ -240,7 +240,7 @@ describe("computeHashlineDiff", () => { // fires through computeHashlineDiff but produces identical content. const text = `${line}\n`; const snapshotStore = new InMemorySnapshotStore(); - const tag = snapshotStore.recordContiguous(sourcePath, 1, text.split("\n"), { fullText: text }); + const tag = snapshotStore.record(sourcePath, text); const input = `${formatHashlineHeader(sourcePath, tag)}\nreplace 1..1:\n+${line}\n`; const result = await computeHashlineDiff({ input }, tempDir, snapshotStore); expect("error" in result).toBe(true); diff --git a/packages/coding-agent/test/read-column-truncation-snapshot.test.ts b/packages/coding-agent/test/read-column-truncation-snapshot.test.ts index 4cd3fdf5b..a01c8113d 100644 --- a/packages/coding-agent/test/read-column-truncation-snapshot.test.ts +++ b/packages/coding-agent/test/read-column-truncation-snapshot.test.ts @@ -23,7 +23,7 @@ import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import type { ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read"; import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; -const HASHLINE_HEADER_LINE = /^¶(\S+)#([0-9A-F]{3})$/m; +const HASHLINE_HEADER_LINE = /^¶(\S+)#([0-9A-F]{4})$/m; const COLUMN_CAP = 64; const LONG_LINE_LEN = COLUMN_CAP * 3; @@ -115,8 +115,8 @@ describe("read tool column truncation vs hashline snapshot", () => { expect(snapshot).not.toBeNull(); // The snapshot MUST hold the on-disk text, not the display-truncated version. - expect(snapshot?.fullText).toBe(fullText); - expect(snapshot?.get(2)).toBe(longLine); + expect(snapshot?.text).toBe(fullText); + expect(snapshot?.text.split("\n")[1]).toBe(longLine); }); it("range read snapshot keeps untruncated content for long lines", async () => { @@ -133,7 +133,7 @@ describe("read tool column truncation vs hashline snapshot", () => { const { tag } = extractHeader(text); const snapshot = getFileSnapshotStore(session).byHash(filePath, tag); - expect(snapshot?.get(2)).toBe(longLine); + expect(snapshot?.text.split("\n")[1]).toBe(longLine); }); it("multi-range read snapshot keeps untruncated content for long lines", async () => { @@ -150,8 +150,8 @@ describe("read tool column truncation vs hashline snapshot", () => { const { tag } = extractHeader(text); const snapshot = getFileSnapshotStore(session).byHash(filePath, tag); - expect(snapshot?.get(2)).toBe(longLine); - expect(snapshot?.get(6)).toBe(longLine); + expect(snapshot?.text.split("\n")[1]).toBe(longLine); + expect(snapshot?.text.split("\n")[5]).toBe(longLine); }); it("edit can apply against a file with long lines without re-reading", async () => { diff --git a/packages/coding-agent/test/tools/ast-edit.test.ts b/packages/coding-agent/test/tools/ast-edit.test.ts index dcb9aacf9..5262909e1 100644 --- a/packages/coding-agent/test/tools/ast-edit.test.ts +++ b/packages/coding-agent/test/tools/ast-edit.test.ts @@ -212,8 +212,8 @@ describe("ast_edit tool schema", () => { | undefined; // Tree-grouped output: `# packages/pkg-…/src/` then `## root.ts# (1 replacement)`. - expect(text).toMatch(/^## root\.ts#[0-9A-F]{3} \(\d+ replacement[s]?\)$/m); - expect(text).toMatch(/^## child\.ts#[0-9A-F]{3} \(\d+ replacement[s]?\)$/m); + expect(text).toMatch(/^## root\.ts#[0-9A-F]{4} \(\d+ replacement[s]?\)$/m); + expect(text).toMatch(/^## child\.ts#[0-9A-F]{4} \(\d+ replacement[s]?\)$/m); expect(text).not.toContain("ignore.js"); expect(text).not.toContain("outside.ts"); expect(details?.totalReplacements).toBe(2); diff --git a/packages/coding-agent/test/tools/ast-grep.test.ts b/packages/coding-agent/test/tools/ast-grep.test.ts index c6ccaca3f..d327f1ba5 100644 --- a/packages/coding-agent/test/tools/ast-grep.test.ts +++ b/packages/coding-agent/test/tools/ast-grep.test.ts @@ -101,8 +101,8 @@ describe("ast_grep parse errors", () => { const details = result.details as { matchCount?: number; fileCount?: number } | undefined; // Directory mode uses tree-grouped `# dir/` + `## name#hash` headers. - expect(text).toMatch(/## root\.ts#[0-9A-F]{3}/); - expect(text).toMatch(/## child\.ts#[0-9A-F]{3}/); +expect(text).toMatch(/## root\.ts#[0-9A-F]{4}/); +expect(text).toMatch(/## child\.ts#[0-9A-F]{4}/); expect(text).not.toContain("ignore.js"); expect(text).not.toContain("outside.ts"); expect(details?.matchCount).toBe(2); diff --git a/packages/coding-agent/test/tools/search-internal-urls.test.ts b/packages/coding-agent/test/tools/search-internal-urls.test.ts index bf288edc5..df95c9ca3 100644 --- a/packages/coding-agent/test/tools/search-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/search-internal-urls.test.ts @@ -147,7 +147,7 @@ describe("SearchTool internal URL resolution", () => { const text = getResultText(result); expect(text).toContain("needle"); // No hashline section headers or numbered editable lines for immutable sources. - expect(text).not.toMatch(/^¶.*#[0-9A-F]{3}$/m); + expect(text).not.toMatch(/^¶.*#[0-9A-F]{4}$/m); expect(text).not.toMatch(/^\*?\s*\d+:/m); }); @@ -187,7 +187,7 @@ describe("SearchTool internal URL resolution", () => { const text = getResultText(result); expect(text).toContain("needle"); // Mutable local:// sources keep a hashline section header plus numbered match lines. - expect(text).toMatch(/^¶.*#[0-9A-F]{3}$/m); + expect(text).toMatch(/^¶.*#[0-9A-F]{4}$/m); expect(text).toMatch(/^\*\d+:.*needle/m); }); @@ -207,7 +207,7 @@ describe("SearchTool internal URL resolution", () => { const text = getResultText(result); expect(text).toContain("needle"); // Mutable mixed.txt keeps hashlines somewhere in the output. - expect(text).toMatch(/^# mixed\.txt#[0-9A-F]{3}/m); + expect(text).toMatch(/^# mixed\.txt#[0-9A-F]{4}/m); expect(text).toMatch(/^\*\d+:.*mixed needle/m); }); diff --git a/packages/coding-agent/test/tools/search-path-lists.test.ts b/packages/coding-agent/test/tools/search-path-lists.test.ts index 0408ea072..837ed5cd2 100644 --- a/packages/coding-agent/test/tools/search-path-lists.test.ts +++ b/packages/coding-agent/test/tools/search-path-lists.test.ts @@ -145,9 +145,9 @@ describe("tool path arrays", () => { const text = getText(result); const details = result.details as { fileCount?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{3}/m); - expect(text).toMatch(/^# packages\/\n## grep\.txt#[0-9A-F]{3}/m); - expect(text).toMatch(/^# phases\/\n## grep\.txt#[0-9A-F]{3}/m); + expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{4}/m); + expect(text).toMatch(/^# packages\/\n## grep\.txt#[0-9A-F]{4}/m); + expect(text).toMatch(/^# phases\/\n## grep\.txt#[0-9A-F]{4}/m); expect(text).toContain("shared-needle"); expect(text).not.toContain("# other"); expect(details?.fileCount).toBe(3); @@ -359,7 +359,7 @@ describe("tool path arrays", () => { const text = getText(result); const details = result.details as { fileCount?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{3}/m); + expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{4}/m); expect(text).toContain("shared-needle"); expect(text).not.toContain(tempDir); expect(details?.fileCount).toBe(1); @@ -416,9 +416,9 @@ describe("tool path arrays", () => { const text = getText(result); const details = result.details as { fileCount?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# apps\/\n## ast\.ts#[0-9A-F]{3}/m); - expect(text).toMatch(/^# packages\/\n## ast\.ts#[0-9A-F]{3}/m); - expect(text).toMatch(/^# phases\/\n## ast\.ts#[0-9A-F]{3}/m); + expect(text).toMatch(/^# apps\/\n## ast\.ts#[0-9A-F]{4}/m); + expect(text).toMatch(/^# packages\/\n## ast\.ts#[0-9A-F]{4}/m); + expect(text).toMatch(/^# phases\/\n## ast\.ts#[0-9A-F]{4}/m); expect(text).not.toContain("# other"); expect(details?.fileCount).toBe(3); expect(details?.scopePath).toBe("apps/**/*.ts, packages/**/*.ts, phases/**/*.ts"); @@ -444,9 +444,9 @@ describe("tool path arrays", () => { const text = getText(preview); const details = preview.details as { totalReplacements?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# apps\/\n## ast\.ts#[0-9A-F]{3} \(\d+ replacement/m); - expect(text).toMatch(/^# packages\/\n## ast\.ts#[0-9A-F]{3} \(\d+ replacement/m); - expect(text).toMatch(/^# phases\/\n## ast\.ts#[0-9A-F]{3} \(\d+ replacement/m); + expect(text).toMatch(/^# apps\/\n## ast\.ts#[0-9A-F]{4} \(\d+ replacement/m); + expect(text).toMatch(/^# packages\/\n## ast\.ts#[0-9A-F]{4} \(\d+ replacement/m); + expect(text).toMatch(/^# phases\/\n## ast\.ts#[0-9A-F]{4} \(\d+ replacement/m); expect(text).not.toContain("# other"); expect(details?.totalReplacements).toBe(3); expect(details?.scopePath).toBe("apps/**/*.ts, packages/**/*.ts, phases/**/*.ts"); @@ -556,9 +556,9 @@ describe("tool path arrays", () => { const text = getText(result); const details = result.details as { fileCount?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{3}/m); - expect(text).toMatch(/^# packages\/\n## grep\.txt#[0-9A-F]{3}/m); - expect(text).toMatch(/^# phases\/\n## grep\.txt#[0-9A-F]{3}/m); + expect(text).toMatch(/^# apps\/\n## grep\.txt#[0-9A-F]{4}/m); + expect(text).toMatch(/^# packages\/\n## grep\.txt#[0-9A-F]{4}/m); + expect(text).toMatch(/^# phases\/\n## grep\.txt#[0-9A-F]{4}/m); expect(text).not.toContain("# other"); expect(details?.fileCount).toBe(3); expect(details?.scopePath).toBe("apps, packages, phases"); @@ -583,8 +583,8 @@ describe("tool path arrays", () => { const text = getText(result); const details = result.details as { fileCount?: number; scopePath?: string } | undefined; - expect(text).toMatch(/^# alpha\.txt#[0-9A-F]{3}/m); - expect(text).toMatch(/^# beta\.txt#[0-9A-F]{3}/m); + expect(text).toMatch(/^# alpha\.txt#[0-9A-F]{4}/m); + expect(text).toMatch(/^# beta\.txt#[0-9A-F]{4}/m); expect(text).toContain("exact-needle alpha"); expect(text).toContain("exact-needle beta"); expect(text).not.toContain("nested"); diff --git a/packages/coding-agent/test/write-hashline-header.test.ts b/packages/coding-agent/test/write-hashline-header.test.ts index 17c84a865..2e1e965d2 100644 --- a/packages/coding-agent/test/write-hashline-header.test.ts +++ b/packages/coding-agent/test/write-hashline-header.test.ts @@ -30,7 +30,7 @@ function resultText(result: { content: { type: string; text?: string }[] }): str .join("\n"); } -const HASHLINE_HEADER_LINE = /^¶(\S+)#([0-9A-F]{3})$/; +const HASHLINE_HEADER_LINE = /^¶(\S+)#([0-9A-F]{4})$/; describe("write tool hashline header", () => { let tmpDir: string; @@ -67,7 +67,7 @@ describe("write tool hashline header", () => { // follow-up edit can land without an extra `read` round-trip. const snapshot = getFileSnapshotStore(session).byHash(filePath, tag!); expect(snapshot).not.toBeNull(); - expect(snapshot?.fullText).toBe(content); + expect(snapshot?.text).toBe(content); }); it("makes the post-write tag usable by the hashline patcher", async () => { diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index e927b2400..ac683d6ad 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -1,15 +1,28 @@ # Changelog ## [Unreleased] - ### Breaking Changes +- Changed hashline section tags from 3-hex to 4-hex content-hash tags, so legacy 3-digit tags are no longer valid - Changed hashline syntax to verb-based v4: body-bearing ops are `replace N..M:`, `insert before N:`, `insert after N:`, `insert head:`, and `insert tail:`, while bodyless `delete N..M` handles deletion. Removed `>A..B` repeat rows and the old `prepend:` / `append:` virtual insert headers; `-` rows remain rejected with a teaching error. ### Added +- Added `maxPaths` and `maxVersionsPerPath` options to `InMemorySnapshotStore` to bound tracked paths and per-path snapshot history - Re-introduced balance-validated boundary repair in `applyEdits`. A replacement hunk (`replace N..M:` + body) is normalized so its payload preserves the deleted region's delimiter balance: when the body restates a closing delimiter that survives just outside the range (duplicate `}` / `);` / `]`) the echo is dropped, and when the range deletes a structural closer the body never restates (missing closer) the closer is spared instead of deleted. A repair fires only when one boundary operation drives the per-channel `()` / `[]` / `{}` imbalance to exactly zero while leaving surrounding text byte-identical (single-line ops are limited to pure structural-closer lines), so balance-preserving edits and intentional balanced duplicates are never touched. Bracket counting skips strings, template literals, and comments. Each repair surfaces a `delimiter-balance` warning through `ApplyResult.warnings`. +### Changed + +- Changed patch application to accept edits whenever the live file's normalized content hash matches the section tag, even when that anchor was not covered by a stored snapshot + +### Removed + +- Removed `SnapshotStore.recordContiguous` and `SnapshotStore.recordSparse` in favor of full-file `record(path, fullText)` snapshots + +### Fixed + +- Fixed hash mismatch rejections caused by CRLF or trailing spaces/tabs by normalizing those characters before computing file-hash tags + ## [15.5.12] - 2026-05-29 ### Changed diff --git a/packages/hashline/bench/recovery-session-chain.ts b/packages/hashline/bench/recovery-session-chain.ts index feb99d330..fb205afa4 100644 --- a/packages/hashline/bench/recovery-session-chain.ts +++ b/packages/hashline/bench/recovery-session-chain.ts @@ -44,8 +44,8 @@ function seed(lines: number, rewrittenLine: number): Fixture { const v0Text = `${v0Lines.join("\n")}\n`; const v1Text = `${v1Lines.join("\n")}\n`; const store = new InMemorySnapshotStore(); - const h0 = store.recordContiguous(PATH, 1, v0Text.split("\n"), { fullText: v0Text }); - store.recordContiguous(PATH, 1, v1Text.split("\n"), { fullText: v1Text }); + const h0 = store.record(PATH, v0Text); + store.record(PATH, v1Text); return { store, v1Text, h0 }; } diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 4817c3d3a..cbf3a8f68 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -33,7 +33,8 @@ "fmt": "biome format --write ." }, "dependencies": { - "diff": "catalog:" + "diff": "catalog:", + "lru-cache": "catalog:" }, "devDependencies": { "@types/bun": "catalog:" diff --git a/packages/hashline/src/format.ts b/packages/hashline/src/format.ts index 28c3e28a2..6867f9e0c 100644 --- a/packages/hashline/src/format.ts +++ b/packages/hashline/src/format.ts @@ -72,23 +72,38 @@ export function formatInsertHeader(cursor: Cursor): string { } } -/** Number of hex characters in an opaque snapshot tag. */ -export const HL_FILE_HASH_LENGTH = 3; - -/** Canonical uppercase hexadecimal opaque snapshot tag carried by a hashline section header. */ +/** Number of hex characters in a content-derived file-hash tag. */ +export const HL_FILE_HASH_LENGTH = 4; +/** Canonical uppercase hexadecimal content-hash tag carried by a hashline section header. */ export const HL_FILE_HASH_RE_RAW = `[0-9A-F]{${HL_FILE_HASH_LENGTH}}`; - /** Capture-group form of {@link HL_FILE_HASH_RE_RAW}. */ export const HL_FILE_HASH_CAPTURE_RE_RAW = `(${HL_FILE_HASH_RE_RAW})`; - /** Regex-escaped form of {@link HL_LINE_BODY_SEP}, safe for embedding inside a regex. */ export const HL_LINE_BODY_SEP_RE_RAW = regexEscape(HL_LINE_BODY_SEP); - /** - * Representative snapshot tags for use in user-facing error messages and + * Representative file-hash tags for use in user-facing error messages and * prompt examples. */ -export const HL_FILE_HASH_EXAMPLES = ["0A3", "1F7", "3C9"] as const; +export const HL_FILE_HASH_EXAMPLES = ["1A2B", "3C4D", "9F3E"] as const; +/** + * Normalize text before hashing: trim trailing `[ \t\r]` from every line (and + * the final line) in a single pass so CRLF endings and display-trimmed lines + * do not invalidate a tag. + */ +function normalizeFileHashText(text: string): string { + return text.replace(/[ \t\r]+(?=\n|$)/g, ""); +} +/** + * Compute the content-derived hash tag carried by a hashline section header. + * The tag is a 4-hex fingerprint of the whole file's normalized text: any read + * of byte-identical content mints the same tag, and a follow-up edit anchored + * at any line validates whenever the live file still hashes to it. + */ +export function computeFileHash(text: string): string { + const normalized = normalizeFileHashText(text); + const low16 = Bun.hash.xxHash32(normalized, 0) & 0xffff; + return low16.toString(16).padStart(HL_FILE_HASH_LENGTH, "0").toUpperCase(); +} /** * Format a comma-separated list of example anchors with an optional line-number diff --git a/packages/hashline/src/grammar.lark b/packages/hashline/src/grammar.lark index b9c7c2e17..73d7e4e50 100644 --- a/packages/hashline/src/grammar.lark +++ b/packages/hashline/src/grammar.lark @@ -4,7 +4,7 @@ end_patch: "*** End Patch" LF? file_patch: file_header hunk+ file_header: "¶" filename ("#" file_hash)? LF -file_hash: /[0-9A-F]{3}/ +file_hash: /[0-9A-F]{4}/ filename: /[^\s#]+/ hunk: body_hunk | delete_hunk diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts index 074e4be62..3f8021314 100644 --- a/packages/hashline/src/input.ts +++ b/packages/hashline/src/input.ts @@ -9,7 +9,7 @@ */ import * as path from "node:path"; import { applyEdits } from "./apply"; -import { HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import { HL_FILE_HASH_LENGTH, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; import { parsePatch, parsePatchStreaming } from "./parser"; import { Tokenizer } from "./tokenizer"; import type { ApplyResult, Edit, SplitOptions } from "./types"; @@ -56,7 +56,7 @@ function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null { if (!line.startsWith(HL_FILE_PREFIX)) return null; const body = stripApplyPatchPathNoise(line.slice(HL_FILE_PREFIX.length).trim()); if (body.length === 0) return null; - const match = /^(\S+?)(?:#([0-9A-Fa-f]{3}))?\s*$/.exec(body); + const match = new RegExp(`^(\\S+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body); if (match === null) return null; const path = normalizeHashlinePath(match[1], cwd); if (path.length === 0) return null; @@ -95,7 +95,7 @@ function parseHashlineHeaderLine(line: string, cwd?: string): RawSection | null const recovered = tryParseRecoveryHeader(trimmed, cwd); if (recovered !== null) return recovered; throw new Error( - `Input header must be ${HL_FILE_PREFIX}PATH or ${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}TAG with a 3-hex snapshot tag; got ${JSON.stringify(trimmed)}.`, + `Input header must be ${HL_FILE_PREFIX}PATH or ${HL_FILE_PREFIX}PATH${HL_FILE_HASH_SEP}TAG with a ${HL_FILE_HASH_LENGTH}-hex content-hash tag; got ${JSON.stringify(trimmed)}.`, ); } diff --git a/packages/hashline/src/mismatch.ts b/packages/hashline/src/mismatch.ts index 6cc554a40..b77d02454 100644 --- a/packages/hashline/src/mismatch.ts +++ b/packages/hashline/src/mismatch.ts @@ -6,17 +6,16 @@ * plus a couple of lines of surrounding context. The {@link MismatchError} * formats this into a message at construction time. */ -import { formatNumberedLine, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import { formatNumberedLine, HL_FILE_HASH_EXAMPLES, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; import { MISMATCH_CONTEXT } from "./messages"; const LINE_REF_RE = /^\s*[>+\-*]*\s*(\d+)(?::.*)?\s*$/; - /** Format the required-shape diagnostic shown when a line reference is malformed. */ export function formatFullAnchorRequirement(raw?: string): string { const received = raw === undefined ? "" : ` Received ${JSON.stringify(raw)}.`; return ( - `a bare line number from read/search output plus the section header snapshot tag ` + - `(for example ${HL_FILE_PREFIX}src/foo.ts${HL_FILE_HASH_SEP}0A3 and line "160")${received}` + `a bare line number from read/search output plus the section header content-hash tag ` + + `(for example ${HL_FILE_PREFIX}src/foo.ts${HL_FILE_HASH_SEP}${HL_FILE_HASH_EXAMPLES[0]} and line "160")${received}` ); } diff --git a/packages/hashline/src/patcher.ts b/packages/hashline/src/patcher.ts index fcc3db688..0aac9335e 100644 --- a/packages/hashline/src/patcher.ts +++ b/packages/hashline/src/patcher.ts @@ -23,14 +23,14 @@ * filesystem configuration. */ import { applyEdits } from "./apply"; -import { formatHashlineHeader, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; +import { computeFileHash, formatHashlineHeader, HL_FILE_HASH_SEP, HL_FILE_PREFIX } from "./format"; import type { Filesystem, WriteResult } from "./fs"; import { isNotFound } from "./fs"; import type { Patch, PatchSection } from "./input"; import { MismatchError } from "./mismatch"; import { detectLineEnding, type LineEnding, normalizeToLF, restoreLineEndings, stripBom } from "./normalize"; import { Recovery, type RecoveryResult } from "./recovery"; -import type { Snapshot, SnapshotStore } from "./snapshots"; +import type { SnapshotStore } from "./snapshots"; import type { ApplyResult, Edit } from "./types"; export interface PatcherOptions { @@ -116,25 +116,6 @@ function recoveryToApplyResult(result: RecoveryResult): ApplyResult { warnings: result.warnings, }; } - -/** - * Decide whether `snapshot` proves the live file is byte-for-byte the read - * the model authored against. Two shapes: - * - Full-text snapshot: cheap string equality. - * - Sparse snapshot (e.g. selector reads, search hits): every anchor line - * must be in the snapshot AND every recorded line must match the live - * file. Without this branch, sparse reads can't short-circuit and fall - * through to recovery, which declines them as "patcher-owned direct - * apply" — yielding a spurious MismatchError on unchanged files. - */ -function snapshotProvesUnchanged(snapshot: Snapshot, currentText: string, section: PatchSection): boolean { - if (snapshot.fullText !== undefined) return snapshot.fullText === currentText; - for (const lineNumber of section.collectAnchorLines()) { - if (snapshot.get(lineNumber) === undefined) return false; - } - return snapshot.matchesLiveFile(currentText.split("\n")); -} - function mergeWarnings(...sources: ReadonlyArray): string[] { const out: string[] = []; for (const source of sources) { @@ -324,9 +305,8 @@ export class Patcher { } #recordFullSnapshot(canonicalPath: string, normalized: string): string { - return this.snapshots.recordContiguous(canonicalPath, 1, normalized.split("\n"), { fullText: normalized }); + return this.snapshots.record(canonicalPath, normalized); } - #applyWithRecovery(args: { section: PatchSection; canonicalPath: string; @@ -337,29 +317,27 @@ export class Patcher { const { section, canonicalPath, exists, normalized, edits } = args; const expected = exists ? section.fileHash : undefined; if (expected === undefined) return applyEdits(normalized, [...edits]); - - const snapshot = this.snapshots.byHash(canonicalPath, expected); - if (snapshot && snapshotProvesUnchanged(snapshot, normalized, section)) { - return applyEdits(normalized, [...edits]); - } - if (snapshot) { - const recovered = this.recovery.tryRecover({ - path: canonicalPath, - currentText: normalized, - fileHash: expected, - edits, - }); - if (recovered) return recoveryToApplyResult(recovered); - } - - const currentHash = this.#recordFullSnapshot(canonicalPath, normalized); + // Whole-file unchanged → the tag still names the live content, so an + // edit anchored at ANY line (displayed or not) is safe to apply. + if (computeFileHash(normalized) === expected) return applyEdits(normalized, [...edits]); + // File drifted: try to replay the edit against the version the tag + // names and 3-way-merge it onto the live content. + const recovered = this.recovery.tryRecover({ + path: canonicalPath, + currentText: normalized, + fileHash: expected, + edits, + }); + if (recovered) return recoveryToApplyResult(recovered); + const hashRecognized = this.snapshots.byHash(canonicalPath, expected) !== null; + const actualFileHash = this.#recordFullSnapshot(canonicalPath, normalized); throw new MismatchError({ path: section.path, expectedFileHash: expected, - actualFileHash: currentHash, + actualFileHash, fileLines: normalized.split("\n"), anchorLines: section.collectAnchorLines(), - hashRecognized: snapshot !== null, + hashRecognized, }); } } diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts index 7bd506652..197b967bc 100644 --- a/packages/hashline/src/recovery.ts +++ b/packages/hashline/src/recovery.ts @@ -128,35 +128,6 @@ function replaySessionChainOnCurrent( }; } -function snapshotHasEntries(snapshot: Snapshot): boolean { - for (const _entry of snapshot.entries()) return true; - return false; -} - -function buildSparseOverlayText(currentText: string, snapshot: Snapshot): string { - const overlaid = currentText.split("\n"); - let maxCachedLine = 0; - for (const [lineNum] of snapshot.entries()) { - if (lineNum > maxCachedLine) maxCachedLine = lineNum; - } - while (overlaid.length < maxCachedLine) overlaid.push(""); - for (const [lineNum, content] of snapshot.entries()) { - overlaid[lineNum - 1] = content; - } - return overlaid.join("\n"); -} - -function sparseSnapshotCoversAnchors(snapshot: Snapshot, edits: readonly Edit[]): boolean { - for (const lineNumber of collectAnchorLines(edits)) { - if (snapshot.get(lineNumber) === undefined) return false; - } - return true; -} - -function sparseSnapshotMatchesCurrent(currentText: string, snapshot: Snapshot): boolean { - return snapshot.matchesLiveFile(currentText.split("\n")); -} - /** First 1-indexed line at which `a` and `b` diverge, or `undefined` if equal. */ function findFirstChangedLine(a: string, b: string): number | undefined { if (a === b) return undefined; @@ -175,52 +146,38 @@ function isHeadSnapshot(head: Snapshot | null, snapshot: Snapshot): boolean { /** * Stateless recovery driver over a {@link SnapshotStore}. Construct once and - * call {@link Recovery.tryRecover} per stale-hash incident. The default - * implementation tries three strategies in order: + * call {@link Recovery.tryRecover} per stale-tag incident. The default + * implementation tries two strategies in order: * - * 1. Apply on the cached `fullText` snapshot, then 3-way-merge onto current. - * 2. (Session chain) If the snapshot wasn't the head, retry on current text - * when line counts match AND every edit's anchor line content is unchanged - * between snapshot and current — the previous in-session edit advanced - * the hash and the model's anchors still name the same logical rows. Emits - * a dedicated {@link RECOVERY_SESSION_REPLAY_WARNING} because even with - * both guards a coincidental insert+delete pair on duplicate rows can - * still land the edit on the wrong row; see {@link replaySessionChainOnCurrent}. - * 3. Reconstruct from a sparse snapshot (lines map only), then 3-way-merge. - * Sparse snapshots that still match the live file are direct-apply cases - * owned by the patcher, so recovery declines them. + * 1. Apply the edits on the full-file version the tag names, then 3-way-merge + * the resulting patch onto the live content (handles external writes). + * 2. (Session chain) If that version wasn't the head, replay the edits onto + * the live content directly when line counts match AND every edit's anchor + * line content is unchanged between version and current — a prior in-session + * edit advanced the tag and the model's anchors still name the same logical + * rows. Emits a dedicated {@link RECOVERY_SESSION_REPLAY_WARNING} because + * even with both guards a coincidental insert+delete pair on duplicate rows + * can still land the edit on the wrong row; see {@link replaySessionChainOnCurrent}. */ export class Recovery { constructor(readonly store: SnapshotStore) {} - /** * Attempt recovery. Returns `null` when no path forward is found — the * caller should then surface a {@link MismatchError}. */ tryRecover(args: RecoveryArgs): RecoveryResult | null { const { path, currentText, fileHash, edits } = args; - const head = this.store.head(path); const snapshot = this.store.byHash(path, fileHash); - if (!snapshot || !snapshotHasEntries(snapshot)) return null; - - const isHead = isHeadSnapshot(head, snapshot); + if (!snapshot) return null; + const isHead = isHeadSnapshot(this.store.head(path), snapshot); const recoveryWarning = isHead ? RECOVERY_EXTERNAL_WARNING : RECOVERY_SESSION_CHAIN_WARNING; - const isSessionChain = !isHead; - - if (snapshot.fullText !== undefined) { - const merged = applyEditsToSnapshot(snapshot.fullText, currentText, edits, recoveryWarning); - if (merged !== null) return merged; - // Session-chain fallback: the 3-way merge on the snapshot refused. - // Replay onto current is gated by line-count equality AND - // anchor-content alignment — see `replaySessionChainOnCurrent` - // for why both guards together still don't fully prove correctness. - if (isSessionChain) return replaySessionChainOnCurrent(snapshot.fullText, currentText, edits); - return null; - } - - if (!sparseSnapshotCoversAnchors(snapshot, edits)) return null; - if (sparseSnapshotMatchesCurrent(currentText, snapshot)) return null; - const overlayText = buildSparseOverlayText(currentText, snapshot); - return applyEditsToSnapshot(overlayText, currentText, edits, recoveryWarning); + const merged = applyEditsToSnapshot(snapshot.text, currentText, edits, recoveryWarning); + if (merged !== null) return merged; + // Session-chain fallback: the 3-way merge on the version refused. + // Replay onto current is gated by line-count equality AND + // anchor-content alignment — see `replaySessionChainOnCurrent` + // for why both guards together still don't fully prove correctness. + if (!isHead) return replaySessionChainOnCurrent(snapshot.text, currentText, edits); + return null; } } diff --git a/packages/hashline/src/snapshots.ts b/packages/hashline/src/snapshots.ts index df4cf011c..179b3e571 100644 --- a/packages/hashline/src/snapshots.ts +++ b/packages/hashline/src/snapshots.ts @@ -1,434 +1,128 @@ /** * Per-session snapshot store used by {@link Recovery} and {@link Patcher} to - * bind hashline section tags to the exact file view that minted them. + * bind hashline section tags to the exact file content that minted them. * - * Producers (typically `read` / `search` tools) record the lines they showed - * the model. The store returns a three-hex opaque tag. Consumers resolve that - * tag back to the recorded snapshot and verify the recorded lines against the - * live file before applying anchored edits. + * A section tag is a content-derived hash of the *whole file* (see + * {@link computeFileHash}). Any read of byte-identical content mints the same + * tag, so reads of one file state fuse onto one anchor and a follow-up edit + * anchored at any line validates whenever the live file still hashes to it. * - * Tags are scrambled by a deterministic permutation of the 12-bit hex space - * built once at module load. Slot index `i` always maps to the same tag - * across restarts (good for tests and reproducibility), but consecutive slot - * indices produce unrelated tags — `mint()` followed by another `mint()` does - * not return e.g. `000` then `001`, and the first tag a store ever hands out - * is not `000`. This is hallucination prevention, not adversarial security: it - * stops an LLM from guessing `001` after observing `000`, or assuming any - * monotonic counter pattern. The patcher catches stale-tag misuse via - * content verification at apply time, so determinism doesn't reduce safety. + * Producers (typically `read` / `search` / `write` tools) call + * {@link SnapshotStore.record} with the full normalized text they observed. + * The store hashes it, dedups against the per-path history, and returns the + * tag. Consumers (the patcher) resolve a stale tag back to the recorded full + * text via {@link SnapshotStore.byHash} and 3-way-merge the would-be edit onto + * the live content. * - * Tag → slot resolution is a single `Map` lookup (no `parseInt`, no regex): - * the inverse table is populated with both lowercase and uppercase forms of - * every tag at module load. - * - * Snapshots are an open abstract type. The two concrete impls cover the - * shapes producers actually emit: {@link ContiguousSnapshot} for `read`-style - * runs (no map allocation, range arithmetic for lookup and superset checks) - * and {@link SparseSnapshot} for `search`-style hits. - * - * The abstract base class lets callers plug in whatever storage they like. - * {@link InMemorySnapshotStore} ships as a single 4096-slot ring shared across - * paths — snapshots carry their own path, so the global ring is fine and - * `byHash` rejects cross-path lookups. + * The abstract base class lets callers plug in whatever storage they like + * (LRU, persistent SQLite, etc.). {@link InMemorySnapshotStore} ships as a + * sensible default backed by `lru-cache`: a bounded set of paths, each with a + * short history of full-file versions so in-session edit chains can still + * recover against the version a stale tag names. */ +import { LRUCache } from "lru-cache/raw"; +import { computeFileHash } from "./format"; /** - * One snapshot of a file view as observed at a point in time. The two - * primitive methods subclasses must supply — `get` and `entries` — give callers - * (and the default `isSuperset` / `matchesLiveFile` impls) everything they - * need to verify recorded content against the live file. + * One full-file version observed at a point in time. The tag the model sees is + * {@link Snapshot.hash}; recovery replays edits against {@link Snapshot.text}. */ -export abstract class Snapshot { - /** Canonical path this snapshot belongs to. */ - abstract readonly path: string; - /** Timestamp (ms since epoch) the snapshot was recorded. */ - abstract readonly recordedAt: number; - /** Full normalized text when the read observed the whole file. */ - abstract readonly fullText?: string; - - /** Recorded content for 1-indexed `lineNumber`, or `undefined` if this snapshot doesn't cover that line. */ - abstract get(lineNumber: number): string | undefined; - - /** Iterate (1-indexed lineNumber, content) pairs in stable order. */ - abstract entries(): Iterable<[number, string]>; - - /** True iff every (line, content) the `other` snapshot asserts is also present here with matching content. */ - isSuperset(other: Snapshot): boolean { - for (const [lineNumber, content] of other.entries()) { - if (this.get(lineNumber) !== content) return false; - } - return true; - } - - /** True iff every recorded line matches `currentLines` (0-indexed array of live file lines). */ - matchesLiveFile(currentLines: readonly string[]): boolean { - for (const [lineNumber, content] of this.entries()) { - if (currentLines[lineNumber - 1] !== content) return false; - } - return true; - } +export interface Snapshot { + /** Canonical path this version belongs to. */ + readonly path: string; + /** Full normalized (LF, no BOM) file text as observed. */ + readonly text: string; + /** Content-derived tag for {@link Snapshot.text} (see {@link computeFileHash}). */ + readonly hash: string; + /** Timestamp (ms since epoch) the version was recorded. */ + recordedAt: number; } /** - * A contiguous run of lines starting at `offset` (1-indexed). Backed by a - * plain `string[]`; lookup is `lines[n - offset]`, superset against another - * contiguous snapshot is pure range arithmetic. - */ -export class ContiguousSnapshot extends Snapshot { - constructor( - public readonly path: string, - public readonly offset: number, - public readonly lines: readonly string[], - public readonly fullText?: string, - public readonly recordedAt: number = Date.now(), - ) { - super(); - } - - get(lineNumber: number): string | undefined { - const index = lineNumber - this.offset; - if (index < 0 || index >= this.lines.length) return undefined; - return this.lines[index]; - } - - *entries(): IterableIterator<[number, string]> { - for (let index = 0; index < this.lines.length; index++) { - yield [this.offset + index, this.lines[index] ?? ""]; - } - } - - override isSuperset(other: Snapshot): boolean { - if (other instanceof ContiguousSnapshot) { - if (other.offset < this.offset) return false; - const skip = other.offset - this.offset; - if (skip + other.lines.length > this.lines.length) return false; - for (let index = 0; index < other.lines.length; index++) { - if (this.lines[skip + index] !== other.lines[index]) return false; - } - return true; - } - return super.isSuperset(other); - } - - override matchesLiveFile(currentLines: readonly string[]): boolean { - for (let index = 0; index < this.lines.length; index++) { - if (currentLines[this.offset + index - 1] !== this.lines[index]) return false; - } - return true; - } -} - -/** - * A sparse `(lineNumber → content)` map, used for snapshots that don't cover a - * single contiguous run — e.g. search hits plus their context windows. - */ -export class SparseSnapshot extends Snapshot { - constructor( - public readonly path: string, - public readonly lines: ReadonlyMap, - public readonly fullText?: string, - public readonly recordedAt: number = Date.now(), - ) { - super(); - } - - get(lineNumber: number): string | undefined { - return this.lines.get(lineNumber); - } - - entries(): IterableIterator<[number, string]> { - return this.lines.entries(); - } -} - -/** Optional metadata supplied at snapshot record time. */ -export interface SnapshotMetadata { - /** Full normalized text, when the producer observed the whole file. */ - fullText?: string; -} - -/** - * Storage seam for file-content snapshots. Hashline section tags are opaque - * store pointers; without the store that minted them they carry no meaning. + * Storage seam for full-file version snapshots. The patcher calls {@link head} + * for the latest version of a path and {@link byHash} when it needs the + * specific historical version a section's stale tag names. */ export abstract class SnapshotStore { - /** Most-recently pushed snapshot for `path`, or `null` if none. */ + /** Most-recently recorded version for `path`, or `null` if none. */ abstract head(path: string): Snapshot | null; - /** Snapshot currently occupying `tag`'s slot for `path`, or `null`. */ - abstract byHash(path: string, tag: string): Snapshot | null; + /** Recorded version for `path` whose tag equals `hash`, or `null`. */ + abstract byHash(path: string, hash: string): Snapshot | null; - /** Record a contiguous run of lines (e.g. from a `read` tool). `startLine` is 1-indexed. */ - abstract recordContiguous( - path: string, - startLine: number, - lines: readonly string[], - metadata?: SnapshotMetadata, - ): string; + /** Record the full normalized text of `path` and return its content tag. */ + abstract record(path: string, fullText: string): string; - /** Record sparse `(lineNumber, content)` pairs (e.g. a `search` match plus context). */ - abstract recordSparse( - path: string, - entries: Iterable, - metadata?: SnapshotMetadata, - ): string; - - /** Drop snapshots belonging to a single path. */ + /** Drop the version history for a single path. */ abstract invalidate(path: string): void; - /** Drop every snapshot. */ + /** Drop every version history. */ abstract clear(): void; } -const RING_SIZE = 0x1000; -const RING_MASK = RING_SIZE - 1; -const FALLBACK_TAG = "000"; -const HEX_DIGITS = "0123456789ABCDEF"; +const DEFAULT_MAX_PATHS = 30; +const DEFAULT_MAX_VERSIONS_PER_PATH = 4; + +export interface InMemorySnapshotStoreOptions { + /** Maximum number of distinct paths tracked at once (default 30). LRU eviction. */ + maxPaths?: number; + /** Maximum full-file versions retained per path (default 4). Oldest dropped first. */ + maxVersionsPerPath?: number; +} /** - * Deterministic permutation of `[0..4095]` → 3-hex tag, plus its inverse. + * In-memory {@link SnapshotStore} backed by `lru-cache`. Per-path history is a + * short ring of full-file versions (oldest dropped first); per-session path + * tracking is LRU-bounded so cold paths age out automatically. * - * Built once at module load via Mulberry32 + Fisher–Yates with a fixed seed. - * Verified properties (see test suite): bijection over the 4096 hex values, - * `FORWARD[0] !== "000"`, no recoverable arithmetic pattern in consecutive - * tags. Cryptographic strength is not required — this only has to defeat - * trivial LLM extrapolation like "after `000` comes `001`". - */ -const { FORWARD, INVERSE } = buildHexTables(); - -function formatSlotTag(value: number): string { - return ( - (HEX_DIGITS[(value >>> 8) & 0xf] ?? "0") + - (HEX_DIGITS[(value >>> 4) & 0xf] ?? "0") + - (HEX_DIGITS[value & 0xf] ?? "0") - ); -} - -function buildHexTables(): { FORWARD: readonly string[]; INVERSE: ReadonlyMap } { - let state = 0x9e3779b9 | 0; - const rng = (): number => { - state = (state + 0x6d2b79f5) | 0; - let t = state; - t = Math.imul(t ^ (t >>> 15), t | 1); - t ^= t + Math.imul(t ^ (t >>> 7), t | 61); - return ((t ^ (t >>> 14)) >>> 0) / 0x100000000; - }; - - const order = Array.from({ length: RING_SIZE }, (_, index) => index); - for (let index = RING_SIZE - 1; index > 0; index--) { - const swapIndex = Math.floor(rng() * (index + 1)); - const tmp = order[index] ?? 0; - order[index] = order[swapIndex] ?? 0; - order[swapIndex] = tmp; - } - - const forward = order.map(formatSlotTag); - const inverse = new Map(); - for (let slot = 0; slot < RING_SIZE; slot++) { - const tag = forward[slot] ?? FALLBACK_TAG; - inverse.set(tag, slot); - inverse.set(tag.toLowerCase(), slot); - } - return { FORWARD: forward, INVERSE: inverse }; -} - -/** - * In-memory {@link SnapshotStore} backed by a flat 4096-slot ring shared across - * all paths. Slot allocation is a simple `counter & 0xfff`; the tag the model - * sees is `FORWARD[slot]` from the module-level permutation, so consecutive - * pushes hand out unrelated tags. Before allocating, {@link InMemorySnapshotStore} - * folds a new view into an existing same-path slot when the two agree on every - * shared line: one covering the other reuses it verbatim (dedup), overlapping - * or abutting runs extend in place, and gapped runs union into a sparse view - * (coalesce). All reuse the original tag, so sequential reads of an unchanged - * file collapse onto one anchor instead of fragmenting. A disagreeing shared - * line means the file changed on disk, so a fresh slot (new tag) is minted. - * Slot reuse on wrap is intentional: stale tags may - * alias after 4096 distinct pushes, and the patcher catches misuse by verifying - * the resolved snapshot's content (and path) against the live file before - * applying edits. + * Recording byte-identical content again refreshes recency and reuses the + * existing tag (read fusion); recording new content unshifts a fresh version + * onto the front of the path history. */ export class InMemorySnapshotStore extends SnapshotStore { - readonly #slots: Array = new Array(RING_SIZE).fill(null); - #nextCounter = 0; - #filled = 0; + readonly #versions: LRUCache; + readonly #maxVersionsPerPath: number; + + constructor(options: InMemorySnapshotStoreOptions = {}) { + super(); + this.#versions = new LRUCache({ max: options.maxPaths ?? DEFAULT_MAX_PATHS }); + this.#maxVersionsPerPath = options.maxVersionsPerPath ?? DEFAULT_MAX_VERSIONS_PER_PATH; + } head(path: string): Snapshot | null { - for (let offset = 1; offset <= this.#filled; offset++) { - const snapshot = this.#slots[(this.#nextCounter - offset) & RING_MASK]; - if (snapshot && snapshot.path === path) return snapshot; + return this.#versions.get(path)?.[0] ?? null; + } + + byHash(path: string, hash: string): Snapshot | null { + const history = this.#versions.get(path); + return history?.find(version => version.hash === hash) ?? null; + } + + record(path: string, fullText: string): string { + const hash = computeFileHash(fullText); + // `get` refreshes LRU recency for `path`. + const history = this.#versions.get(path) ?? []; + const existing = history.find(version => version.hash === hash); + if (existing) { + // Same content state observed again: refresh recency and promote to + // head (it is the current file content), then reuse the tag. + existing.recordedAt = Date.now(); + if (history[0] !== existing) { + this.#versions.set(path, [existing, ...history.filter(version => version !== existing)]); + } + return hash; } - return null; - } - byHash(path: string, tag: string): Snapshot | null { - const slot = INVERSE.get(tag); - if (slot === undefined) return null; - const snapshot = this.#slots[slot]; - if (!snapshot || snapshot.path !== path) return null; - return snapshot; - } - - recordContiguous( - path: string, - startLine: number, - lines: readonly string[], - metadata: SnapshotMetadata = {}, - ): string { - return this.#record(new ContiguousSnapshot(path, startLine, lines, metadata.fullText)); - } - - recordSparse(path: string, entries: Iterable, metadata: SnapshotMetadata = {}): string { - const lines = new Map(); - for (const [lineNumber, content] of entries) lines.set(lineNumber, content); - return this.#record(new SparseSnapshot(path, lines, metadata.fullText)); + const snapshot: Snapshot = { path, text: fullText, hash, recordedAt: Date.now() }; + this.#versions.set(path, [snapshot, ...history].slice(0, this.#maxVersionsPerPath)); + return hash; } invalidate(path: string): void { - for (let index = 0; index < RING_SIZE; index++) { - if (this.#slots[index]?.path === path) this.#slots[index] = null; - } + this.#versions.delete(path); } clear(): void { - this.#slots.fill(null); - } - - #record(incoming: Snapshot): string { - // Walk newest→oldest for a same-path snapshot we can fold `incoming` - // into: either it already covers `incoming` (dedup) or the two agree on - // every shared line and merge into one run/sparse view (coalesce). - // Folding keeps the original slot, so the tag the model already saw for - // an earlier read also anchors this one. - for (let offset = 1; offset <= this.#filled; offset++) { - const slot = (this.#nextCounter - offset) & RING_MASK; - const existing = this.#slots[slot]; - if (!existing || existing.path !== incoming.path) continue; - const folded = coalesceSnapshots(existing, incoming); - if (folded === null) continue; - if (folded !== existing) this.#slots[slot] = folded; - return FORWARD[slot] ?? FALLBACK_TAG; - } - - const slot = this.#nextCounter & RING_MASK; - this.#slots[slot] = incoming; - this.#nextCounter++; - if (this.#filled < RING_SIZE) this.#filled++; - return FORWARD[slot] ?? FALLBACK_TAG; + this.#versions.clear(); } } - -/** - * Fold `incoming` into `existing` (callers guarantee same path). Returns: - * - `existing` when it already covers every line `incoming` asserts — pure - * dedup, no new storage required; - * - a fresh merged snapshot when the two agree on every shared line — a - * {@link ContiguousSnapshot} when the union is a single run (overlapping or - * abutting reads), otherwise a {@link SparseSnapshot} spanning the gap(s). - * Agreement is the "file unchanged" proof, so one tag can anchor both; - * - `null` when a shared line disagrees: the file changed on disk between the - * reads, so the views describe different states and MUST keep distinct tags. - * - * Disjoint reads share no lines and so never conflict — they union optimistically - * (the patcher re-verifies recorded lines against live content before applying, - * so a stale union degrades to a re-read prompt, never a corrupt edit). - */ -function coalesceSnapshots(existing: Snapshot, incoming: Snapshot): Snapshot | null { - // Contiguous∩contiguous is the hot path (sequential range reads); settle it - // with range arithmetic so dedup and in-run extension allocate nothing. - if ( - existing instanceof ContiguousSnapshot && - incoming instanceof ContiguousSnapshot && - existing.lines.length > 0 && - incoming.lines.length > 0 - ) { - return coalesceContiguous(existing, incoming); - } - return coalesceGeneral(existing, incoming); -} - -/** Range-arithmetic coalesce for two non-empty contiguous runs. */ -function coalesceContiguous(a: ContiguousSnapshot, b: ContiguousSnapshot): Snapshot | null { - const aEnd = a.offset + a.lines.length - 1; - const bEnd = b.offset + b.lines.length - 1; - - // Every shared line must agree, else the file changed between the reads. - const lo = Math.max(a.offset, b.offset); - const hi = Math.min(aEnd, bEnd); - for (let line = lo; line <= hi; line++) { - if (a.lines[line - a.offset] !== b.lines[line - b.offset]) return null; - } - - // `a` already covers `b` verbatim → reuse the slot untouched. - if (b.offset >= a.offset && bEnd <= aEnd) return a; - - // Overlapping or directly abutting → a single, larger contiguous run. - if (b.offset <= aEnd + 1 && a.offset <= bEnd + 1) { - const start = Math.min(a.offset, b.offset); - const end = Math.max(aEnd, bEnd); - const lines = new Array(end - start + 1); - for (let i = 0; i < a.lines.length; i++) lines[a.offset - start + i] = a.lines[i] ?? ""; - // `b` is the fresher read; overlay it last (shared lines are equal anyway). - for (let i = 0; i < b.lines.length; i++) lines[b.offset - start + i] = b.lines[i] ?? ""; - return new ContiguousSnapshot(a.path, start, lines, pickFullText(a, b, start, lines)); - } - - // A gap separates the runs → fold into a sparse view that preserves it. - return unionSnapshots(a, b); -} - -/** Entry-based coalesce covering any snapshot shape (sparse, or mixed runs). */ -function coalesceGeneral(existing: Snapshot, incoming: Snapshot): Snapshot | null { - let covered = 0; - let total = 0; - for (const [line, content] of incoming.entries()) { - total++; - const seen = existing.get(line); - if (seen === undefined) continue; - if (seen !== content) return null; - covered++; - } - if (covered === total) return existing; - return unionSnapshots(existing, incoming); -} - -/** - * Union two compatible views (callers guarantee agreement on shared lines). - * Collapses back to a {@link ContiguousSnapshot} when the merged line numbers - * form a gap-free run, otherwise yields a {@link SparseSnapshot}. - */ -function unionSnapshots(a: Snapshot, b: Snapshot): Snapshot { - const merged = new Map(); - for (const [line, content] of a.entries()) merged.set(line, content); - // `b` is the fresher read; it wins ties (shared lines are equal anyway). - for (const [line, content] of b.entries()) merged.set(line, content); - - let min = Number.POSITIVE_INFINITY; - let max = Number.NEGATIVE_INFINITY; - for (const line of merged.keys()) { - if (line < min) min = line; - if (line > max) max = line; - } - - if (max - min + 1 === merged.size) { - const lines = new Array(merged.size); - for (const [line, content] of merged) lines[line - min] = content; - return new ContiguousSnapshot(a.path, min, lines, pickFullText(a, b, min, lines)); - } - - const ordered = [...merged].sort((x, y) => x[0] - y[0]); - return new SparseSnapshot(a.path, new Map(ordered)); -} - -/** - * Carry a whole-file `fullText` onto a merged run only when it is provably - * still accurate: the run must start at line 1 and reconstruct the candidate - * byte-for-byte. Otherwise the text is stale (the file grew past it) and the - * snapshot falls back to line-by-line verification. - */ -function pickFullText(a: Snapshot, b: Snapshot, start: number, lines: readonly string[]): string | undefined { - if (start !== 1) return undefined; - const candidate = b.fullText ?? a.fullText; - if (candidate === undefined) return undefined; - return candidate === lines.join("\n") ? candidate : undefined; -} diff --git a/packages/hashline/test/boundary-repair.test.ts b/packages/hashline/test/boundary-repair.test.ts index 0928837fd..9d62bae0c 100644 --- a/packages/hashline/test/boundary-repair.test.ts +++ b/packages/hashline/test/boundary-repair.test.ts @@ -146,7 +146,7 @@ describe("boundary-balance repair through stale-snapshot recovery", () => { const currentText = snapshotText.replace("const tail = 0;", "const tail = 99;"); const store = new InMemorySnapshotStore(); - const fileHash = store.recordContiguous(PATH, 1, snapshotText.split("\n"), { fullText: snapshotText }); + const fileHash = store.record(PATH, snapshotText); // `replace 4..5:` replaces the body lines but the payload also restates the `});` // that survives at line 6 — the duplicate-closer mistake. diff --git a/packages/hashline/test/patcher.test.ts b/packages/hashline/test/patcher.test.ts index 58c42fd62..aa15f0065 100644 --- a/packages/hashline/test/patcher.test.ts +++ b/packages/hashline/test/patcher.test.ts @@ -1,5 +1,12 @@ import { describe, expect, it } from "bun:test"; -import { InMemoryFilesystem, InMemorySnapshotStore, MismatchError, Patch, Patcher } from "@oh-my-pi/hashline"; +import { + computeFileHash, + InMemoryFilesystem, + InMemorySnapshotStore, + MismatchError, + Patch, + Patcher, +} from "@oh-my-pi/hashline"; const PATH = "a.ts"; @@ -11,47 +18,49 @@ describe("Patcher snapshot tag integrity", () => { expect(() => new Patcher(options)).toThrow(/requires a SnapshotStore/); }); - it("applies when the section tag resolves to a matching snapshot", async () => { + it("applies when the section tag is the live file's content hash", async () => { const fs = new InMemoryFilesystem([[PATH, "before\n"]]); const snapshots = new InMemorySnapshotStore(); - const tag = snapshots.recordContiguous(PATH, 1, ["before", ""], { fullText: "before\n" }); + const tag = snapshots.record(PATH, "before\n"); const patcher = new Patcher({ fs, snapshots }); const result = await patcher.apply(Patch.parse(`¶${PATH}#${tag}\nreplace 1..1:\n+after`)); expect(result.sections[0]?.op).toBe("update"); - expect(result.sections[0]?.fileHash).toMatch(/^[0-9A-F]{3}$/); + expect(result.sections[0]?.fileHash).toMatch(/^[0-9A-F]{4}$/); expect(result.sections[0]?.fileHash).not.toBe(tag); expect(fs.get(PATH)).toBe("after\n"); }); - it("normalizes lowercase section tags while parsing", () => { - const section = Patch.parseSingle(`¶${PATH}#0a3\nreplace 1..1:\n+after`); - - expect(section.fileHash).toBe("0A3"); - }); - - it("rejects a wrapped tag whose slot now holds unrelated content", async () => { - const fs = new InMemoryFilesystem([[PATH, "target\n"]]); + it("validates any anchor purely from the content hash, even with no recorded snapshot", async () => { + // The core fix: the tag fingerprints the WHOLE file. An edit anchored at + // a line the model never saw recorded applies whenever the live file + // still hashes to the tag — no stored snapshot is consulted. + const content = "l1\nl2\nl3\nl4\nl5\n"; + const fs = new InMemoryFilesystem([[PATH, content]]); const snapshots = new InMemorySnapshotStore(); - for (let index = 0; index < 10; index++) { - snapshots.recordContiguous(PATH, 1, [`warmup ${index}`]); - } - const staleTag = snapshots.recordContiguous(PATH, 1, ["target", ""], { fullText: "target\n" }); - for (let index = 0; index < 4096; index++) { - snapshots.recordContiguous(PATH, 1, [`unrelated ${index}`]); - } + const tag = computeFileHash(content); + // Store is intentionally empty: byHash(tag) === null. + expect(snapshots.byHash(PATH, tag)).toBeNull(); const patcher = new Patcher({ fs, snapshots }); - const patch = Patch.parse(`¶${PATH}#${staleTag}\nreplace 1..1:\n+changed`); - await expect(patcher.apply(patch)).rejects.toBeInstanceOf(MismatchError); - expect(fs.get(PATH)).toBe("target\n"); + const result = await patcher.apply(Patch.parse(`¶${PATH}#${tag}\nreplace 3..3:\n+L3`)); + + expect(result.sections[0]?.op).toBe("update"); + expect(fs.get(PATH)).toBe("l1\nl2\nL3\nl4\nl5\n"); }); - it("refuses with mismatch when the snapshot exists for the hash but content drifted", async () => { + it("normalizes lowercase section tags while parsing", () => { + const section = Patch.parseSingle(`¶${PATH}#1a2b\nreplace 1..1:\n+after`); + + expect(section.fileHash).toBe("1A2B"); + }); + + it("refuses with mismatch when the recorded version no longer matches live content", async () => { const fs = new InMemoryFilesystem([[PATH, "drifted\n"]]); const snapshots = new InMemorySnapshotStore(); - const tag = snapshots.recordContiguous(PATH, 1, ["before", ""], { fullText: "before\n" }); + // Tag was minted from "before\n" but the live file is "drifted\n". + const tag = snapshots.record(PATH, "before\n"); const patcher = new Patcher({ fs, snapshots }); try { @@ -68,24 +77,26 @@ describe("Patcher snapshot tag integrity", () => { expect(fs.get(PATH)).toBe("drifted\n"); }); - it("refuses with a 'not from this session' diagnostic when the hash was never recorded for this path", async () => { + it("refuses with a 'not from this session' diagnostic when the tag was never recorded for this path", async () => { const fs = new InMemoryFilesystem([[PATH, "current\n"]]); const snapshots = new InMemorySnapshotStore(); const patcher = new Patcher({ fs, snapshots }); - // `#FFF` parses cleanly as a 3-hex slot tag but no snapshot has ever - // been minted into that slot for this path — equivalent to the model - // either fabricating the hash or carrying it over from a prior session. + // A 4-hex tag that is neither the live content hash nor a recorded + // version — equivalent to the model fabricating it or carrying it over + // from a prior session. + const live = computeFileHash("current\n"); + const bogus = live === "FFFF" ? "0000" : "FFFF"; try { - await patcher.apply(Patch.parse(`¶${PATH}#FFF\nreplace 1..1:\n+after`)); + await patcher.apply(Patch.parse(`¶${PATH}#${bogus}\nreplace 1..1:\n+after`)); throw new Error("expected MismatchError"); } catch (error) { expect(error).toBeInstanceOf(MismatchError); const message = (error as MismatchError).displayMessage; - expect(message).toMatch(/hash #FFF is not from this session/); + expect(message).toMatch(new RegExp(`hash #${bogus} is not from this session`)); expect(message).toMatch(/never invent the tag/); // Still surfaces the current hash so the model can pivot to a re-read. - expect(message).toMatch(/current file hashes to #[0-9A-F]{3}/); + expect(message).toMatch(/current file hashes to #[0-9A-F]{4}/); } expect(fs.get(PATH)).toBe("current\n"); }); diff --git a/packages/hashline/test/recovery-session-chain.test.ts b/packages/hashline/test/recovery-session-chain.test.ts index c24beaf11..24bf947a2 100644 --- a/packages/hashline/test/recovery-session-chain.test.ts +++ b/packages/hashline/test/recovery-session-chain.test.ts @@ -21,8 +21,8 @@ function seedTwoSnapshots(): { store: InMemorySnapshotStore; v0Text: string; v1T v1Lines[4] = "L5-CHANGED"; const v0Text = `${v0Lines.join("\n")}\n`; const v1Text = `${v1Lines.join("\n")}\n`; - const h0 = store.recordContiguous(PATH, 1, v0Text.split("\n"), { fullText: v0Text }); - const h1 = store.recordContiguous(PATH, 1, v1Text.split("\n"), { fullText: v1Text }); + const h0 = store.record(PATH, v0Text); + const h1 = store.record(PATH, v1Text); return { store, v0Text, v1Text, h0, h1 }; } diff --git a/packages/hashline/test/snapshots.test.ts b/packages/hashline/test/snapshots.test.ts index a2672734c..ddd6b23f6 100644 --- a/packages/hashline/test/snapshots.test.ts +++ b/packages/hashline/test/snapshots.test.ts @@ -1,115 +1,88 @@ import { describe, expect, it } from "bun:test"; -import { InMemorySnapshotStore } from "@oh-my-pi/hashline"; +import { computeFileHash, InMemorySnapshotStore } from "@oh-my-pi/hashline"; const PATH = "/tmp/__hashline-snapshots__.ts"; -const TAG_RE = /^[0-9A-F]{3}$/; - -function nextHex(tag: string): string { - return ((Number.parseInt(tag, 16) + 1) & 0xfff).toString(16).toUpperCase().padStart(3, "0"); -} +const OTHER = "/tmp/__hashline-other__.ts"; +const TAG_RE = /^[0-9A-F]{4}$/; describe("InMemorySnapshotStore", () => { - it("reuses a prior tag when that snapshot is a content-matching superset", () => { + it("derives the tag from whole-file content (matches computeFileHash)", () => { const store = new InMemorySnapshotStore(); - const tag = store.recordContiguous(PATH, 1, ["L1", "L2", "L3"]); - + const text = "L1\nL2\nL3\n"; + const tag = store.record(PATH, text); expect(tag).toMatch(TAG_RE); - expect(store.recordContiguous(PATH, 2, ["L2"])).toBe(tag); - expect(store.recordSparse(PATH, [[3, "L3"]])).toBe(tag); + expect(tag).toBe(computeFileHash(text)); }); - it("coalesces overlapping consistent reads into one tag spanning the union", () => { + it("fuses repeated reads of identical content onto one tag", () => { const store = new InMemorySnapshotStore(); - const file = Array.from({ length: 200 }, (_, index) => `L${index + 1}`); - // read a.ts:50-100 then a.ts:100-200 — share line 100, file unchanged. - const first = store.recordContiguous(PATH, 50, file.slice(49, 100)); - const second = store.recordContiguous(PATH, 100, file.slice(99, 200)); - - expect(first).toMatch(TAG_RE); + const text = "alpha\nbeta\ngamma\n"; + const first = store.record(PATH, text); + const second = store.record(PATH, text); expect(second).toBe(first); - const snap = store.byHash(PATH, first); - expect(snap?.get(50)).toBe("L50"); - expect(snap?.get(100)).toBe("L100"); - expect(snap?.get(150)).toBe("L150"); - expect(snap?.get(200)).toBe("L200"); - expect(snap?.get(201)).toBeUndefined(); + // One head, byHash resolves to the same full text. + expect(store.head(PATH)?.hash).toBe(first); + expect(store.byHash(PATH, first)?.text).toBe(text); }); - it("coalesces non-contiguous reads into a sparse snapshot under one tag", () => { + it("mints a new tag when content changes and retains the prior version", () => { const store = new InMemorySnapshotStore(); - const file = Array.from({ length: 200 }, (_, index) => `L${index + 1}`); - // read a.ts:1-100 then a.ts:150-200 — disjoint, gap 101..149. - const first = store.recordContiguous(PATH, 1, file.slice(0, 100)); - const second = store.recordContiguous(PATH, 150, file.slice(149, 200)); - - expect(second).toBe(first); - const snap = store.byHash(PATH, first); - expect(snap?.get(1)).toBe("L1"); - expect(snap?.get(100)).toBe("L100"); - expect(snap?.get(125)).toBeUndefined(); - expect(snap?.get(150)).toBe("L150"); - expect(snap?.get(200)).toBe("L200"); + const v1 = "one\ntwo\n"; + const v2 = "one\ntwo\nthree\n"; + const tag1 = store.record(PATH, v1); + const tag2 = store.record(PATH, v2); + expect(tag2).not.toBe(tag1); + // Head is the latest; the older version is still resolvable by its tag. + expect(store.head(PATH)?.hash).toBe(tag2); + expect(store.byHash(PATH, tag1)?.text).toBe(v1); + expect(store.byHash(PATH, tag2)?.text).toBe(v2); }); - it("mints a new tag when a shared line disagrees, then dedups against it", () => { + it("promotes a re-observed older version back to head", () => { const store = new InMemorySnapshotStore(); - const first = store.recordContiguous(PATH, 50, ["L50", "L51", "L52"]); - // re-read 52-54 but line 52 drifted — the file changed on disk. - const second = store.recordContiguous(PATH, 52, ["CHANGED", "L53", "L54"]); - - expect(second).not.toBe(first); - expect(store.byHash(PATH, first)?.get(52)).toBe("L52"); - expect(store.byHash(PATH, second)?.get(52)).toBe("CHANGED"); - // A follow-up read consistent with the newer view dedups onto its tag. - expect(store.recordContiguous(PATH, 53, ["L53"])).toBe(second); + const v1 = "x\n"; + const v2 = "y\n"; + const tag1 = store.record(PATH, v1); + store.record(PATH, v2); + // File reverts to v1 content: recording it again makes v1 the head. + expect(store.record(PATH, v1)).toBe(tag1); + expect(store.head(PATH)?.hash).toBe(tag1); }); - it("scrambles slot tags so the first and next tags are not predictable counters", () => { - const store = new InMemorySnapshotStore(); - const first = store.recordContiguous(PATH, 1, ["value 0"]); - const second = store.recordContiguous(PATH, 1, ["value 1"]); - - expect(first).toMatch(TAG_RE); - expect(second).toMatch(TAG_RE); - expect(first).not.toBe("000"); - expect(second).not.toBe(nextHex(first)); + it("bounds per-path history to maxVersionsPerPath (oldest dropped)", () => { + const store = new InMemorySnapshotStore({ maxVersionsPerPath: 2 }); + const tagA = store.record(PATH, "A\n"); + const tagB = store.record(PATH, "B\n"); + const tagC = store.record(PATH, "C\n"); + // Only the two newest versions survive. + expect(store.byHash(PATH, tagC)?.text).toBe("C\n"); + expect(store.byHash(PATH, tagB)?.text).toBe("B\n"); + expect(store.byHash(PATH, tagA)).toBeNull(); }); - it("pushes new views into distinct ring slots", () => { - const store = new InMemorySnapshotStore(); - const first = store.recordContiguous(PATH, 1, ["one"]); - const second = store.recordContiguous(PATH, 1, ["two"]); - - expect(first).toMatch(TAG_RE); - expect(second).toMatch(TAG_RE); - expect(second).not.toBe(first); - expect(store.head(PATH)?.get(1)).toBe("two"); - expect(store.byHash(PATH, first)?.get(1)).toBe("one"); - expect(store.byHash(PATH, second)?.get(1)).toBe("two"); + it("bounds tracked paths to maxPaths (cold path evicted)", () => { + const store = new InMemorySnapshotStore({ maxPaths: 1 }); + const tag = store.record(PATH, "first\n"); + store.record(OTHER, "second\n"); + // Recording OTHER evicted PATH from the LRU. + expect(store.byHash(PATH, tag)).toBeNull(); + expect(store.head(PATH)).toBeNull(); }); - it("rejects cross-path lookups even when the tag slot is occupied", () => { + it("rejects cross-path lookups", () => { const store = new InMemorySnapshotStore(); - const tag = store.recordContiguous(PATH, 1, ["one"]); - - expect(store.byHash("/tmp/other.ts", tag)).toBeNull(); + const tag = store.record(PATH, "shared\n"); + expect(store.byHash(OTHER, tag)).toBeNull(); }); - it("wraps after 4096 pushes and byHash returns the new slot occupant", () => { + it("invalidate drops one path; clear drops everything", () => { const store = new InMemorySnapshotStore(); - const first = store.recordContiguous(PATH, 1, ["value 0"]); - let previous = first; - - for (let index = 1; index < 4096; index++) { - const tag = store.recordContiguous(PATH, 1, [`value ${index}`]); - expect(tag).toMatch(TAG_RE); - expect(tag).not.toBe(previous); - previous = tag; - } - const wrapped = store.recordContiguous(PATH, 1, ["value 4096"]); - - expect(wrapped).toBe(first); - expect(store.byHash(PATH, first)?.get(1)).toBe("value 4096"); - expect(store.byHash(PATH, previous)?.get(1)).toBe("value 4095"); + const tagA = store.record(PATH, "A\n"); + const tagB = store.record(OTHER, "B\n"); + store.invalidate(PATH); + expect(store.byHash(PATH, tagA)).toBeNull(); + expect(store.byHash(OTHER, tagB)?.text).toBe("B\n"); + store.clear(); + expect(store.byHash(OTHER, tagB)).toBeNull(); }); }); diff --git a/scripts/session-stats/sync.py b/scripts/session-stats/sync.py index eb38531b8..026373ec8 100644 --- a/scripts/session-stats/sync.py +++ b/scripts/session-stats/sync.py @@ -56,7 +56,7 @@ SCHEMA_VERSION = 3 # Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors # / looks_successful / extract_warnings semantics change. Bump invalidates # previously-stored ss_edit_* rows on next sync. -EDIT_PARSER_VERSION = 5 +EDIT_PARSER_VERSION = 6 SCHEMA_SQL = """ CREATE TABLE IF NOT EXISTS ss_sessions ( @@ -229,31 +229,48 @@ def batch_count_tokens(strings: list[str]) -> list[int]: # --------------------------------------------------------------------------- # # Hashline edit parser. # -# Supports two on-the-wire formats so the analytic tables stay coherent across -# the format transition: +# The session corpus spans several hashline generations, so the parser +# recognizes all of them and normalizes every `¶`/`§` section into the same +# EditSection shape. A `¶` section commits to a grammar from its FIRST op line, +# so the verb and sigil grammars never cross-contaminate (a body line such as +# `delete 5` inside a sigil-era section stays payload, not a phantom delete op). # -# new (current): ¶PATH[#HASH], LINE↑[body], LINE↓[body], A[-B]:[body], A[-B]! -# legacy: §PATH, «ANCHOR, »ANCHOR, ≔ANCHOR[..ANCHOR] +# verb (current): ¶PATH[#TAG] replace N..M: / delete N..M / +# insert before N: / insert after N: / +# insert head: / insert tail: (+ `+TEXT` body rows) +# sigil (corpus): ¶PATH[#TAG] N↑[body] / N↓[body] / A[-B]:[body] / A[-B]! +# legacy: §PATH «ANCHOR / »ANCHOR / ≔ANCHOR[..ANCHOR] # -# Legacy "anchor" tokens were `<2-letter-hash>` (e.g. `4fb`, `12*`); -# new ops use bare line numbers and hoist the file hash into the header. - -_LEGACY_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$") -_LEGACY_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$") -_LEGACY_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$") +# TAG width/case drifted across releases (2-4 hex, lower or upper, sometimes +# absent), so the header accepts any `#` suffix instead of a fixed width. +# Legacy "anchor" tokens were `<2-letter-hash>` (e.g. `4fb`, `12*`). # Header: one or more `¶`, optional whitespace, path (no whitespace/#/¶), -# optional `#HASH` (4 lowercase hex). -_HEADER_NEW_RE = re.compile(r"^¶+\s*([^\s#¶]+)(?:#([0-9a-f]{4}))?\s*$") +# optional `#TAG` of any width/case. +_HEADER_NEW_RE = re.compile(r"^¶+\s*([^\s#¶]+)(?:#\S+)?\s*$") + +# Verb-based v4 (current) ops; body rows are `+TEXT` on the following lines. +_VERB_REPLACE_RE = re.compile(r"^\s*replace\s+([1-9][0-9]*)(?:\s*(?:\.\.|-|…)\s*([1-9][0-9]*))?\s*:?\s*$") +_VERB_DELETE_RE = re.compile(r"^\s*delete\s+([1-9][0-9]*)(?:\s*(?:\.\.|-|…)\s*([1-9][0-9]*))?\s*$") +_VERB_INSERT_RE = re.compile( + r"^\s*insert\s+(?:(?Pbefore|after)\s+(?P[1-9][0-9]*)|(?Phead|tail))\s*:?\s*$" +) + +# Sigil/colon ops (historical corpus); body rows are bare lines. # Insert op: LINE↑BODY / LINE↓BODY / BOF↑BODY / EOF↓BODY … -_OP_INSERT_NEW_RE = re.compile( +_OP_INSERT_HL_RE = re.compile( r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*|BOF|EOF)(?P[↑↓])(?P.*)$" ) # Replace / delete op: A:BODY / A-B:BODY / A! / A-B! -_OP_RANGE_NEW_RE = re.compile( +_OP_RANGE_HL_RE = re.compile( r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*)(?:-(?P[1-9][0-9]*))?(?P[:!])(?P.*)$" ) +# Legacy `§`/`«»≔` ops. +_LEGACY_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$") +_LEGACY_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$") +_LEGACY_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$") + _HASHLINE_ENVELOPE_MARKERS = {"*** Begin Patch", "*** End Patch", "*** Abort"} @@ -306,8 +323,9 @@ class EditSection: def parse_hashline_input(input_str: str) -> list[EditSection]: sections: list[EditSection] = [] cur: EditSection | None = None - cur_format: str | None = None # "new" | "legacy" - open_idx: int | None = None # current open payload block in cur + cur_format: str | None = None # "hash" (¶) | "legacy" (§) + cur_grammar: str | None = None # within "hash": None | "verb" | "sigil" + open_idx: int | None = None # current open payload block in cur def open_new(s: EditSection) -> int: s.payload_blocks.append([]) @@ -322,13 +340,14 @@ def parse_hashline_input(input_str: str) -> list[EditSection]: break continue - # Headers — new format first, then legacy. + # Headers — `¶` (verb/sigil eras) first, then legacy `§`. new_header = _HEADER_NEW_RE.match(line) if new_header: if cur is not None: sections.append(cur) cur = EditSection(target_file=new_header.group(1)) - cur_format = "new" + cur_format = "hash" + cur_grammar = None open_idx = None continue if line.startswith("§"): @@ -339,15 +358,65 @@ def parse_hashline_input(input_str: str) -> list[EditSection]: prefix_end += 1 cur = EditSection(target_file=line[prefix_end:].strip()) cur_format = "legacy" + cur_grammar = None open_idx = None continue if cur is None: continue - if cur_format == "new": - ins = _OP_INSERT_NEW_RE.match(line) + if cur_format == "hash": + # Verb-based v4 ops; tried only while the grammar is undecided or + # already verb, so sigil-era body lines never match a verb keyword. + if cur_grammar in (None, "verb"): + m = _VERB_REPLACE_RE.match(line) + if m: + cur_grammar = "verb" + a = int(m.group(1)) + b = int(m.group(2)) if m.group(2) else a + cur.deleted_lines += max(b - a + 1, 1) + cur.op_anchors.append(str(a)) + if b != a: + cur.op_anchors.append(str(b)) + cur.touch(a) + cur.touch(b) + cur.op_count += 1 + open_idx = open_new(cur) + continue + m = _VERB_DELETE_RE.match(line) + if m: + cur_grammar = "verb" + a = int(m.group(1)) + b = int(m.group(2)) if m.group(2) else a + cur.deleted_lines += max(b - a + 1, 1) + cur.op_anchors.append(str(a)) + if b != a: + cur.op_anchors.append(str(b)) + cur.touch(a) + cur.touch(b) + cur.op_count += 1 + open_idx = None # delete carries no body + continue + m = _VERB_INSERT_RE.match(line) + if m: + cur_grammar = "verb" + anchor = m.group("anchor") + if anchor is not None: + cur.op_anchors.append(anchor) + cur.touch(int(anchor)) + cur.op_count += 1 + open_idx = open_new(cur) + continue + if cur_grammar == "verb": + # Body rows are `+TEXT` (`+` alone = blank line); skip stray rows. + if open_idx is not None and line.startswith("+"): + cur.payload_blocks[open_idx].append(line[1:]) + continue + + # Sigil/colon ops (historical corpus); body rows are bare lines. + ins = _OP_INSERT_HL_RE.match(line) if ins: + cur_grammar = "sigil" anchor = ins.group("anchor") inline = ins.group("inline") cur.op_anchors.append(anchor) @@ -362,8 +431,9 @@ def parse_hashline_input(input_str: str) -> list[EditSection]: cur.payload_blocks[open_idx].append(inline) continue - rng = _OP_RANGE_NEW_RE.match(line) + rng = _OP_RANGE_HL_RE.match(line) if rng: + cur_grammar = "sigil" sigil = rng.group("sigil") a_str = rng.group("a") b_str = rng.group("b") or a_str