fix(hashline): rejected malformed snapshot tag suffixes

Aligned the runtime header parser with the formal grammar's no-`#`-in-filename rule: any `#` left in the path body after detecting the trailing `#XXXX` tag means the header is malformed (short `#1A2`, non-hex `#1A2G`, over-long `#1A2B5`, stale-tag copy-paste, line-suffixed tags), not a path with an embedded hash. Mirrored the same rule on the apply_patch recovery path.

Added strict and recovery regression coverage for the three malformed-tag shapes the reviewer named.
This commit is contained in:
roboomp
2026-06-06 01:43:50 +00:00
parent 7e42c281f8
commit 7fcdfd048f
3 changed files with 25 additions and 25 deletions
+6 -6
View File
@@ -72,12 +72,12 @@ function tryParseRecoveryHeader(line: string, cwd?: string): RawSection | null {
pathText = body.replace(/\s+$/, "");
}
// Same anti-junk rule as the strict tokenizer: a `#XXXX` token followed
// by whitespace+content or a line suffix inside the path body is a
// malformed header (e.g. stale-tag copy-paste like
// `src/a.ts#1A2B copied from read` or `src/a.ts#1A2B:42`), not a path
// with an embedded hex fragment.
if (new RegExp(`#[0-9A-Fa-f]{${HL_FILE_HASH_LENGTH}}(?:\\s|:)`).test(pathText)) return null;
// Same rule as the strict tokenizer: the hashline header grammar uses
// `#` as the path/tag separator and does not allow `#` inside
// filenames. Anything `#` left in the path body — short tags, non-hex
// tags, over-long tags, stale-tag copy-paste, line-suffixed tags —
// means the header is malformed, not a path with an embedded hash.
if (pathText.includes("#")) return null;
const path = normalizeHashlinePath(pathText, cwd);
if (path.length === 0) return null;
+9 -19
View File
@@ -334,25 +334,15 @@ function tryParseHeader(line: string): { path: string; fileHash?: string } | nul
}
}
// Reject stale-tag copy-paste such as `¶src/a.ts#1A2B copied from read`
// and line-suffixed tags such as `¶src/a.ts#1A2B:42`: a `#XXXX`
// token followed by tag-tail junk in the path body is a malformed header,
// not a path-with-embedded-hash. Surface the focused diagnostic instead
// of silently mis-routing the edit.
for (let i = FILE_PREFIX_LENGTH; i + HL_FILE_HASH_LENGTH < pathEnd; i++) {
if (line.charCodeAt(i) !== CHAR_HASH) continue;
let allHex = true;
for (let k = 1; k <= HL_FILE_HASH_LENGTH; k++) {
if (!isHexDigitCode(line.charCodeAt(i + k))) {
allHex = false;
break;
}
}
if (!allHex) continue;
const after = i + HL_FILE_HASH_LENGTH + 1;
if (after < pathEnd && (isWhitespaceCode(line.charCodeAt(after)) || line.charCodeAt(after) === CHAR_COLON)) {
return null;
}
// The hashline header grammar uses `#` as the path/tag separator and
// does not allow `#` inside filenames. Anything `#` left in the path
// body — short tags (`#1A2`), non-hex tags (`#1A2G`), over-long tags
// (`#1A2B5`), stale-tag copy-paste (`#1A2B copied from read`), or
// line-suffixed tags (`#1A2B:42`) — means the header is malformed.
// Surface the focused diagnostic instead of silently mis-routing the
// edit or reporting a missing tag downstream.
for (let i = FILE_PREFIX_LENGTH; i < pathEnd; i++) {
if (line.charCodeAt(i) === CHAR_HASH) return null;
}
if (pathEnd === FILE_PREFIX_LENGTH) return null;
+10
View File
@@ -39,6 +39,16 @@ describe("hashline section headers", () => {
/Input header must be/,
);
});
it("rejects malformed snapshot tags", () => {
expect(() => Patch.parse("¶src/a.ts#1A2\nreplace 1..1:\n+after")).toThrow(/Input header must be/);
expect(() => Patch.parse("¶src/a.ts#1A2G\nreplace 1..1:\n+after")).toThrow(/Input header must be/);
expect(() => Patch.parse("¶src/a.ts#1A2B5\nreplace 1..1:\n+after")).toThrow(/Input header must be/);
});
it("rejects malformed snapshot tags even with apply_patch noise", () => {
expect(() => Patch.parse("¶Update File: src/a.ts#1A2G\nreplace 1..1:\n+after")).toThrow(/Input header must be/);
});
});
describe("hashline core — verb header forms", () => {