diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts index a281093aa..4b43c1e82 100644 --- a/packages/hashline/src/input.ts +++ b/packages/hashline/src/input.ts @@ -72,12 +72,12 @@ function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null { pathText = body.replace(/\s+$/, ""); } - // Same anti-junk rule as the strict tokenizer: a `#XXXX` token followed - // by whitespace+content or a line suffix inside the path body is a - // malformed header (e.g. stale-tag copy-paste like - // `src/a.ts#1A2B copied from read` or `src/a.ts#1A2B:42`), not a path - // with an embedded hex fragment. - if (new RegExp(`#[0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}(?:\\s|:)`).test(pathText)) return null; + // Same rule as the strict tokenizer: the hashline header grammar uses + // `#` as the path/tag separator and does not allow `#` inside + // filenames. Anything `#` left in the path body — short tags, non-hex + // tags, over-long tags, stale-tag copy-paste, line-suffixed tags — + // means the header is malformed, not a path with an embedded hash. + if (pathText.includes("#")) return null; const path = normalizeHashlinePath(pathText, cwd); if (path.length === 0) return null; diff --git a/packages/hashline/src/tokenizer.ts b/packages/hashline/src/tokenizer.ts index 7458f6061..aac2519b7 100644 --- a/packages/hashline/src/tokenizer.ts +++ b/packages/hashline/src/tokenizer.ts @@ -334,25 +334,15 @@ function tryParseHeader(line: string): { path: string; fileHash?: string } | nul } } - // Reject stale-tag copy-paste such as `¶src/a.ts#1A2B copied from read` - // and line-suffixed tags such as `¶src/a.ts#1A2B:42`: a `#XXXX` - // token followed by tag-tail junk in the path body is a malformed header, - // not a path-with-embedded-hash. Surface the focused diagnostic instead - // of silently mis-routing the edit. - for (let i = FILE_PREFIX_LENGTH; i + HL_FILE_HASH_LENGTH < pathEnd; i++) { - if (line.charCodeAt(i) !== CHAR_HASH) continue; - let allHex = true; - for (let k = 1; k <= HL_FILE_HASH_LENGTH; k++) { - if (!isHexDigitCode(line.charCodeAt(i + k))) { - allHex = false; - break; - } - } - if (!allHex) continue; - const after = i + HL_FILE_HASH_LENGTH + 1; - if (after < pathEnd && (isWhitespaceCode(line.charCodeAt(after)) || line.charCodeAt(after) === CHAR_COLON)) { - return null; - } + // The hashline header grammar uses `#` as the path/tag separator and + // does not allow `#` inside filenames. Anything `#` left in the path + // body — short tags (`#1A2`), non-hex tags (`#1A2G`), over-long tags + // (`#1A2B5`), stale-tag copy-paste (`#1A2B copied from read`), or + // line-suffixed tags (`#1A2B:42`) — means the header is malformed. + // Surface the focused diagnostic instead of silently mis-routing the + // edit or reporting a missing tag downstream. + for (let i = FILE_PREFIX_LENGTH; i < pathEnd; i++) { + if (line.charCodeAt(i) === CHAR_HASH) return null; } if (pathEnd === FILE_PREFIX_LENGTH) return null; diff --git a/packages/hashline/test/leniency.test.ts b/packages/hashline/test/leniency.test.ts index b74f68618..362b09f77 100644 --- a/packages/hashline/test/leniency.test.ts +++ b/packages/hashline/test/leniency.test.ts @@ -39,6 +39,16 @@ describe("hashline section headers", () => { /Input header must be/, ); }); + + it("rejects malformed snapshot tags", () => { + expect(() => Patch.parse("¶src/a.ts#1A2\nreplace 1..1:\n+after")).toThrow(/Input header must be/); + expect(() => Patch.parse("¶src/a.ts#1A2G\nreplace 1..1:\n+after")).toThrow(/Input header must be/); + expect(() => Patch.parse("¶src/a.ts#1A2B5\nreplace 1..1:\n+after")).toThrow(/Input header must be/); + }); + + it("rejects malformed snapshot tags even with apply_patch noise", () => { + expect(() => Patch.parse("¶Update File: src/a.ts#1A2G\nreplace 1..1:\n+after")).toThrow(/Input header must be/); + }); }); describe("hashline core — verb header forms", () => {