diff --git a/packages/hashline/src/grammar.lark b/packages/hashline/src/grammar.lark index 8a056d4e1..d2e016ecb 100644 --- a/packages/hashline/src/grammar.lark +++ b/packages/hashline/src/grammar.lark @@ -5,7 +5,7 @@ end_patch: "*** End Patch" LF? file_patch: file_header hunk+ file_header: "¶" filename "#" file_hash LF file_hash: /[0-9A-F]{4}/ -filename: /[^\s#]+/ +filename: /[^#\r\n]+/ hunk: replace_hunk | replace_block_hunk | insert_hunk | delete_hunk | delete_block_hunk replace_hunk: replace_anchor LF emit_op* diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts index 9864d95b6..5b5774b5e 100644 --- a/packages/hashline/src/input.ts +++ b/packages/hashline/src/input.ts @@ -50,14 +50,14 @@ function stripApplyPatchPathNoise(pathText: string): string { * Best-effort recovery for `¶`-prefixed lines the strict tokenizer * rejects. Strips apply_patch keyword noise (`Update File:`, `Update:`, * etc.) and an extra leading `***` (some models emit a hybrid `¶***foo.ts` - * shape), then expects `PATH(#HASH)?` with no embedded whitespace. + * shape), then expects `PATH(#HASH)?`. * Returns `null` when no clean path can be salvaged. */ function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null { if (!line.startsWith(HL_FILE_PREFIX)) return null; const body = stripApplyPatchPathNoise(line.slice(HL_FILE_PREFIX.length).trim()); if (body.length === 0) return null; - const match = new RegExp(`^(\\S+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body); + const match = new RegExp(`^(.+?)(?:#([0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}))?\\s*$`).exec(body); if (match === null) return null; const path = normalizeHashlinePath(match[1], cwd); if (path.length === 0) return null; diff --git a/packages/hashline/src/tokenizer.ts b/packages/hashline/src/tokenizer.ts index 19ffe7164..4bbfbe80c 100644 --- a/packages/hashline/src/tokenizer.ts +++ b/packages/hashline/src/tokenizer.ts @@ -312,28 +312,22 @@ function tryParseHunkHeader(line: string): ParsedHunkHeader | null { function tryParseHeader(line: string): { path: string; fileHash?: string } | null { if (!line.startsWith(HL_FILE_PREFIX)) return null; const end = trimEndIndex(line); - let index = FILE_PREFIX_LENGTH; - if (index >= end) return null; - const pathStart = index; - while (index < end) { - const code = line.charCodeAt(index); - if (code === CHAR_HASH || code === CHAR_SPACE || code === CHAR_TAB) break; - index++; - } - if (index === pathStart) return null; - const path = line.slice(pathStart, index); + if (FILE_PREFIX_LENGTH >= end) return null; + + let pathEnd = end; let fileHash: string | undefined; - if (index < end && line.charCodeAt(index) === CHAR_HASH) { - const hashStart = index + 1; - const hashEnd = hashStart + HL_FILE_HASH_LENGTH; - if (hashEnd > end) return null; - for (let probe = hashStart; probe < hashEnd; probe++) { + const hashStart = end - HL_FILE_HASH_LENGTH - 1; + if (hashStart >= FILE_PREFIX_LENGTH && line.charCodeAt(hashStart) === CHAR_HASH) { + const tagStart = hashStart + 1; + for (let probe = tagStart; probe < end; probe++) { if (!isHexDigitCode(line.charCodeAt(probe))) return null; } - fileHash = line.slice(hashStart, hashEnd).toUpperCase(); - index = hashEnd; + pathEnd = hashStart; + fileHash = line.slice(tagStart, end).toUpperCase(); } - if (skipWhitespace(line, index, end) !== end) return null; + + if (pathEnd === FILE_PREFIX_LENGTH) return null; + const path = line.slice(FILE_PREFIX_LENGTH, pathEnd); return fileHash !== undefined ? { path, fileHash } : { path }; } diff --git a/packages/hashline/test/leniency.test.ts b/packages/hashline/test/leniency.test.ts index f2f2bb2aa..be8298fda 100644 --- a/packages/hashline/test/leniency.test.ts +++ b/packages/hashline/test/leniency.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { applyEdits, parsePatch } from "@oh-my-pi/hashline"; +import { applyEdits, Patch, parsePatch } from "@oh-my-pi/hashline"; function applyPatch(text: string, diff: string): string { return applyEdits(text, parsePatch(diff).edits).text; @@ -7,6 +7,24 @@ function applyPatch(text: string, diff: string): string { const FILE = "a\nb\nc\nd\ne"; +describe("hashline section headers", () => { + it("accepts paths with spaces in anchored section headers", () => { + const section = Patch.parseSingle("¶dir with spaces/file.ts#1a2b\nreplace 1..1:\n+after"); + + expect(section.path).toBe("dir with spaces/file.ts"); + expect(section.fileHash).toBe("1A2B"); + expect(section.applyTo("before").text).toBe("after"); + }); + + it("recovers apply_patch-contaminated headers whose paths contain spaces", () => { + const section = Patch.parseSingle("¶*** Update File: dir with spaces/file.ts#1A2B\nreplace 1..1:\n+after"); + + expect(section.path).toBe("dir with spaces/file.ts"); + expect(section.fileHash).toBe("1A2B"); + expect(section.applyTo("before").text).toBe("after"); + }); +}); + describe("hashline core — verb header forms", () => { it("rejects a bare single-number hunk header with verb guidance", () => { expect(() => parsePatch("2\n+B")).toThrow(/hunk headers need a verb/);