From cc43defbf83e1ad4d1e002f44e0f04b0414b62be Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 3 Jun 2026 07:42:05 +0000 Subject: [PATCH] fix(hashline): accepted spaces in edit paths Parsed hashline section headers by recognizing only a trailing #TAG as the snapshot delimiter, allowing whitespace inside valid paths. Updated recovery parsing and grammar docs to match the runtime parser, and added regression coverage for canonical and recovered headers with spaces. Fixes #1634 --- packages/hashline/src/grammar.lark | 2 +- packages/hashline/src/input.ts | 4 ++-- packages/hashline/src/tokenizer.ts | 30 ++++++++++--------------- packages/hashline/test/leniency.test.ts | 20 ++++++++++++++++- 4 files changed, 34 insertions(+), 22 deletions(-) diff --git a/packages/hashline/src/grammar.lark b/packages/hashline/src/grammar.lark index 8a056d4e1..d2e016ecb 100644 --- a/packages/hashline/src/grammar.lark +++ b/packages/hashline/src/grammar.lark @@ -5,7 +5,7 @@ end_patch: "*** End Patch" LF? file_patch: file_header hunk+ file_header: "¶" filename "#" file_hash LF file_hash: /[0-9A-F]{4}/ -filename: /[^\s#]+/ +filename: /[^#\r\n]+/ hunk: replace_hunk | replace_block_hunk | insert_hunk | delete_hunk | delete_block_hunk replace_hunk: replace_anchor LF emit_op* diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts index 9864d95b6..5b5774b5e 100644 --- a/packages/hashline/src/input.ts +++ b/packages/hashline/src/input.ts @@ -50,14 +50,14 @@ function stripApplyPatchPathNoise(pathText: string): string { * Best-effort recovery for `¶`-prefixed lines the strict tokenizer * rejects. Strips apply_patch keyword noise (`Update File:`, `Update:`, * etc.) and an extra leading `***` (some models emit a hybrid `¶***foo.ts` - * shape), then expects `PATH(#HASH)?` with no embedded whitespace. + * shape), then expects `PATH(#HASH)?`. * Returns `null` when no clean path can be salvaged. */ function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null { if (!line.startsWith(HL_FILE_PREFIX)) return null; const body = stripApplyPatchPathNoise(line.slice(HL_FILE_PREFIX.length).trim()); if (body.length === 0) return null; - const match = new RegExp(`^(\\S+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body); + const match = new RegExp(`^(.+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body); if (match === null) return null; const path = normalizeHashlinePath(match[1], cwd); if (path.length === 0) return null; diff --git a/packages/hashline/src/tokenizer.ts b/packages/hashline/src/tokenizer.ts index 19ffe7164..4bbfbe80c 100644 --- a/packages/hashline/src/tokenizer.ts +++ b/packages/hashline/src/tokenizer.ts @@ -312,28 +312,22 @@ function tryParseHunkHeader(line: string): ParsedHunkHeader | null { function tryParseHeader(line: string): { path: string; fileHash?: string } | null { if (!line.startsWith(HL_FILE_PREFIX)) return null; const end = trimEndIndex(line); - let index = FILE_PREFIX_LENGTH; - if (index >= end) return null; - const pathStart = index; - while (index < end) { - const code = line.charCodeAt(index); - if (code === CHAR_HASH || code === CHAR_SPACE || code === CHAR_TAB) break; - index++; - } - if (index === pathStart) return null; - const path = line.slice(pathStart, index); + if (FILE_PREFIX_LENGTH >= end) return null; + + let pathEnd = end; let fileHash: string | undefined; - if (index < end && line.charCodeAt(index) === CHAR_HASH) { - const hashStart = index + 1; - const hashEnd = hashStart + HL_FILE_HASH_LENGTH; - if (hashEnd > end) return null; - for (let probe = hashStart; probe < hashEnd; probe++) { + const hashStart = end - HL_FILE_HASH_LENGTH - 1; + if (hashStart >= FILE_PREFIX_LENGTH && line.charCodeAt(hashStart) === CHAR_HASH) { + const tagStart = hashStart + 1; + for (let probe = tagStart; probe < end; probe++) { if (!isHexDigitCode(line.charCodeAt(probe))) return null; } - fileHash = line.slice(hashStart, hashEnd).toUpperCase(); - index = hashEnd; + pathEnd = hashStart; + fileHash = line.slice(tagStart, end).toUpperCase(); } - if (skipWhitespace(line, index, end) !== end) return null; + + if (pathEnd === FILE_PREFIX_LENGTH) return null; + const path = line.slice(FILE_PREFIX_LENGTH, pathEnd); return fileHash !== undefined ? { path, fileHash } : { path }; } diff --git a/packages/hashline/test/leniency.test.ts b/packages/hashline/test/leniency.test.ts index f2f2bb2aa..be8298fda 100644 --- a/packages/hashline/test/leniency.test.ts +++ b/packages/hashline/test/leniency.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { applyEdits, parsePatch } from "@oh-my-pi/hashline"; +import { applyEdits, Patch, parsePatch } from "@oh-my-pi/hashline"; function applyPatch(text: string, diff: string): string { return applyEdits(text, parsePatch(diff).edits).text; @@ -7,6 +7,24 @@ function applyPatch(text: string, diff: string): string { const FILE = "a\nb\nc\nd\ne"; +describe("hashline section headers", () => { + it("accepts paths with spaces in anchored section headers", () => { + const section = Patch.parseSingle("¶dir with spaces/file.ts#1a2b\nreplace 1..1:\n+after"); + + expect(section.path).toBe("dir with spaces/file.ts"); + expect(section.fileHash).toBe("1A2B"); + expect(section.applyTo("before").text).toBe("after"); + }); + + it("recovers apply_patch-contaminated headers whose paths contain spaces", () => { + const section = Patch.parseSingle("¶*** Update File: dir with spaces/file.ts#1A2B\nreplace 1..1:\n+after"); + + expect(section.path).toBe("dir with spaces/file.ts"); + expect(section.fileHash).toBe("1A2B"); + expect(section.applyTo("before").text).toBe("after"); + }); +}); + describe("hashline core — verb header forms", () => { it("rejects a bare single-number hunk header with verb guidance", () => { expect(() => parsePatch("2\n+B")).toThrow(/hunk headers need a verb/);