From 91d15b2ec8a3642013e67064b23bed70728eb998 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 28 May 2026 10:21:43 +0200 Subject: [PATCH] fix(hashline)!: removed single-number hunk header shorthand - Rejected bare `A` anchors; single-line ranges must now be spelled `A A`. - Added a descriptive error for single-number headers to guide model output. - Updated grammar, tokenizer, prompt docs, and tests to reflect the change. --- docs/tools/edit.md | 7 +- .../coding-agent/test/core/hashline.test.ts | 7 +- packages/hashline/CHANGELOG.md | 1 + packages/hashline/src/grammar.lark | 2 +- packages/hashline/src/parser.ts | 9 +- packages/hashline/src/prompt.md | 110 ++++++++---------- packages/hashline/src/tokenizer.ts | 48 ++++---- packages/hashline/test/leniency.test.ts | 8 +- 8 files changed, 88 insertions(+), 104 deletions(-) diff --git a/docs/tools/edit.md b/docs/tools/edit.md index 9128ce5e3..a27fe9742 100644 --- a/docs/tools/edit.md +++ b/docs/tools/edit.md @@ -27,7 +27,7 @@ Patch language inside `input`: - **File header**: `¶PATH#TAG` (or `¶PATH` for new-file / virtual-only hunks). `TAG` is three uppercase-hex chars minted by the session snapshot store. -- **Hunk header**: bare `A B` selects original lines A..B. The range separator is normally whitespace; the parser also silently accepts `A-B`, `A..B`, and `A…B` (unicode ellipsis). Virtual variants `BOF` and `EOF` target positions before line 1 / after the last line. The bare single-line shorthand `A` is accepted as `A A`. +- **Hunk header**: bare `A B` selects original lines A..B. Two numbers are REQUIRED — single-line ranges are written `A A` (`5 5`), not `5`. The range separator is normally whitespace; the parser also silently accepts `A-B`, `A..B`, and `A…B` (unicode ellipsis). Virtual variants `BOF` and `EOF` target positions before line 1 / after the last line. - **Body rows** (one per line, immediately under the hunk header): - `+TEXT` — add the literal line `TEXT` verbatim, including all leading whitespace. - `+` alone — add one blank line. @@ -43,7 +43,7 @@ Anchors come from `read`/`search` output. `read` emits a `¶PATH#TAG` header fro Because models reproduce nearby shapes (`read` output, `apply_patch` envelopes, unified-diff hunks), the parser is liberal about a handful of harmless variants: -- `A` — accepted as `A A` (single-line shorthand). +- `A` (bare single number) — REJECTED. The parser throws `single-number hunk header "A" is no longer accepted`. Spell single-line ranges as `A A`. - `A-B`, `A..B`, `A…B` — accepted as `A B` (any of hyphen, double-dot, or unicode ellipsis works as a silent separator). - `&A` — accepted as `&A..A`. - Bare body rows with no `+`/`&` prefix are auto-prepended with `+` and a `BARE_BODY_AUTO_PIPED_WARNING` is appended, BUT only when every row in that block is uniformly bare. Mixed `+`/raw blocks still throw. @@ -165,7 +165,8 @@ Multi-file: - apply_patch / unified-diff contamination: - `line N: apply_patch sentinel "*** …" is not valid in hashline. File sections start with \`¶path#HASH\` (no \`Update File:\` / \`Add File:\` keyword). Hunks are bare \`A B\` lines with \`+TEXT\` / \`&A..B\` body rows.` - `line N: unified-diff hunk header (\`@@ -N,M +N,M @@\`) is not valid in hashline. Hashline hunks are bare \`A B\` lines (or \`BOF\` / \`EOF\` keywords).` - - `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write the range directly: \`5 7\` (or \`5\` for a single line, \`BOF\` / \`EOF\` for virtual positions).` + - `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write the range directly: \`5 7\` (\`BOF\` / \`EOF\` for virtual positions).` + - `line N: single-number hunk header "N" is no longer accepted. Spell single-line ranges as \`N N\` (two numbers); hashline hunks are bare \`A B\` lines (or \`BOF\` / \`EOF\`).` - Out-of-range anchor: - `Line N does not exist (file has M lines)` - Stale snapshot tag throws `MismatchError`. The error contains re-read guidance and nearby current file lines as `*LINE:TEXT` / ` LINE:TEXT`. diff --git a/packages/coding-agent/test/core/hashline.test.ts b/packages/coding-agent/test/core/hashline.test.ts index a70bbcdf3..36f5ee304 100644 --- a/packages/coding-agent/test/core/hashline.test.ts +++ b/packages/coding-agent/test/core/hashline.test.ts @@ -233,12 +233,9 @@ describe("hashline parser — range-anchor syntax", () => { expect(applyDiff(content, range)).toBe("aaa\nBBB\nCCC"); }); - it("accepts bare `A=` as a shorthand for `A..A=`", () => { + it("rejects bare single-number hunk headers (shorthand removed)", () => { const anchor = tag(2, "bbb"); - // `LINE:` is the exact shape `read` renders each file row as, so the - // parser leniently treats it as `LINE-LINE:` for models that - // reproduce the read-output shape as an anchor. - expect(applyDiff(content, `${anchor}\n${repl("BBB")}`)).toBe("aaa\nBBB\nccc"); + expect(() => parseHashline(`${anchor}\n${repl("BBB")}`)).toThrow(/single-number hunk header/); }); it("empty anchor body deletes the range entirely", () => { diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index 0b66eb5cc..d8f5de9af 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -3,6 +3,7 @@ ## [Unreleased] ### Breaking Changes +- Removed the single-number hunk header shorthand. A hunk header now REQUIRES two line numbers (`A A` for a single line, `A B` for a range); a bare `A` row throws `single-number hunk header "A" is no longer accepted`. The `&A` body-row shorthand for `&A..A` is unchanged. - Changed hunk header syntax from `A-B:` to `@@ A..B @@` with `@@ A @@` shorthand for single lines - Changed repeat payload sigil from `^A-B` to `&A..B` with `&A` shorthand for single lines - Changed range separator from `-` to `..` in all contexts (anchors and repeats) diff --git a/packages/hashline/src/grammar.lark b/packages/hashline/src/grammar.lark index b32f79ef1..72b0f50fb 100644 --- a/packages/hashline/src/grammar.lark +++ b/packages/hashline/src/grammar.lark @@ -14,7 +14,7 @@ emit_op: "+" /(.*)/ LF repeat_op: "&" body_range LF anchor: header_range | "BOF" | "EOF" -header_range: LID (WS LID)? +header_range: LID WS LID body_range: LID (".." LID)? LID: /[1-9]\d*/ WS: /[ \t]+/ diff --git a/packages/hashline/src/parser.ts b/packages/hashline/src/parser.ts index adb25720e..8ffaa3781 100644 --- a/packages/hashline/src/parser.ts +++ b/packages/hashline/src/parser.ts @@ -98,7 +98,14 @@ function detectApplyPatchContamination(text: string, _hasPending: boolean): stri const preview = trimmed.length > 48 ? `${trimmed.slice(0, 48)}…` : trimmed; return ( `\`@@\`-bracketed hunk header ${JSON.stringify(preview)} is not valid in hashline. ` + - "Drop the `@@ ... @@` brackets and write the range directly: `5 7` (or `5` for a single line, `BOF` / `EOF` for virtual positions)." + "Drop the `@@ ... @@` brackets and write the range directly: `5 7` (`BOF` / `EOF` for virtual positions)." + ); + } + if (/^[1-9]\d*\s*$/.test(trimmed)) { + return ( + `single-number hunk header ${JSON.stringify(trimmed)} is no longer accepted. ` + + `Spell single-line ranges as \`${trimmed} ${trimmed}\` (two numbers); ` + + "hashline hunks are bare `A B` lines (or `BOF` / `EOF`)." ); } return null; diff --git a/packages/hashline/src/prompt.md b/packages/hashline/src/prompt.md index 2eb7ff8d9..4db0121a4 100644 --- a/packages/hashline/src/prompt.md +++ b/packages/hashline/src/prompt.md @@ -2,45 +2,15 @@ Your patch language selects ranges of file lines and rewrites them. Each hunk pi Every body row is **exactly one** of two kinds: - +TEXT add a new literal line `TEXT` (verbatim, leading whitespace included) - &A..B keep original lines A..B as-is - -`+` and `&` are siblings, not stackable. Never write `+&…`. A row starts with one of them, never both. + &A..B copy lines A..B from snapshot - -This is the original file (the exact shape `read` returns): -``` -¶greet.ts#0A3 -1:export function greet(name: string): string { -2: return `Hello, ${name}!`; -3:} -``` - -To add a null check between the signature and the return, open a hunk on lines 1..3 and list its new content: -``` -¶greet.ts#0A3 -1 3 -&1 -+ if (!name) return "Hello, stranger!"; -&2..3 -``` - -The body says: keep line 1, then add the new literal line, then keep lines 2..3. Result: -``` -1:export function greet(name: string): string { -2: if (!name) return "Hello, stranger!"; -3: return `Hello, ${name}!`; -4:} -``` - - ``` A B select lines A..B; the body rows below describe their new content - (empty body = delete the range) -A select single line A (shorthand for `A A`) + (empty body = delete the range). Always TWO numbers — single + lines are spelled `A A`. BOF virtual position before line 1; body rows insert there EOF virtual position after the last line; body rows insert there ``` @@ -53,52 +23,66 @@ Every file section starts with `¶PATH#HASH`. `HASH` is the snapshot tag from yo -- Anchors are line **numbers**, never line **content**. `read` shows each file row as `LINE:TEXT`; for a patch the hunk header is `4` (or `4 4`) and the body is `+TEXT` (or `&4` to keep it). +- Anchors are line **numbers**, never line **content**, and always come in PAIRS. `read` shows each file row as `LINE:TEXT`; for a patch the hunk header is `4 4` (single line) or `4 7` (range), and the body is `+TEXT` (or `&4` to keep it). +- A bare single number (`4`) is REJECTED — always write two numbers. +- `A B` describes the **original** lines you are replacing. Replacing one line with ten new lines is still `4 4`, NOT `4 13`. - Each range may appear in only ONE hunk per patch. - Line numbers refer to the ORIGINAL file and stay valid for the whole patch — they do not shift as your hunks land. - An empty body **deletes** the selected range entirely. To replace lines A..B with completely new content, list the new content under the hunk header (do not write `&A..B` for the lines you are replacing). - `@@` is NOT a hashline construct. Do not wrap headers in `@@ ... @@` — write the anchor bare. - -# Replace line 1 of `greet.ts#0A3` with two new lines. + + +This is the original file (the exact shape `read` returns): ``` -¶greet.ts#0A3 -1 -+const X = "b"; -+export const Y = X; +¶greet.py#A1 +1:def greet(name): +2: msg = "Hello, " + name +3: print(msg) +4:greet("world") ``` -# Delete lines 2..3 of `greet.ts#0A3`. +# To insert a guard as the first line of greet: ``` -¶greet.ts#0A3 -2 3 +¶greet.py#A1 +1 1 +&1 ++ if not name: name = "stranger" ``` -# Prepend a header. +# Replace line 2 with two new lines. ``` -¶greet.ts#0A3 +2 2 ++ greeting = "Hi" ++ msg = f"{greeting}, {name}" +``` + +# Delete line 4. +``` +¶greet.py#A1 +4 4 +``` + +# Add header & trailer. +``` +¶greet.py#A1 BOF -+// generated header ++# generated header +EOF ++greet("everyone") ``` - + -# WRONG — do not include old lines. -2 3 -- print "hello" -+ print "hi" +# WRONG — range set based on what it will be (RIGHT: 1 1, inserted line count doesn't matter) +1 2 ++def greet(name): ++ """Greet a user by name.""" -# WRONG — do not include context lines. -2 3 - fn hi(): -+ print "hi" - -# WRONG — no `@@` brackets in hashline. -@@ 2..3 @@ -+ print "hi" - -# RIGHT — same intent, well-formed. -2 3 -+ print "hi" +# WRONG — do not include context lines, nor delete old lines, the selector `2 2` itself deletes the entire range +3 3 + msg = "Hello, " + name +- print(msg) ++ return msg diff --git a/packages/hashline/src/tokenizer.ts b/packages/hashline/src/tokenizer.ts index f9ce36101..b7da44639 100644 --- a/packages/hashline/src/tokenizer.ts +++ b/packages/hashline/src/tokenizer.ts @@ -164,8 +164,9 @@ interface RangeScan { * Scan a numeric range for a hunk header. Canonical form is `A B` (two * numbers separated by whitespace); models also reflexively emit `A-B`, * `A..B`, and `A…B` (unicode ellipsis), so we accept any of those as the - * range separator. Bare `A` is the single-line shorthand for `A A`. - * Repeat-row bodies (`&A..B`) keep their own parser; see + * range separator. A second number is REQUIRED — bare `A` is not a valid + * hunk header in this grammar. Repeat-row bodies (`&A..B`) keep their own + * parser and still accept the `&A` single-line shorthand; see * {@link tryParseRepeatPayload}. */ function scanHeaderRange(line: string, index = 0, end = trimEndIndex(line)): RangeScan | null { @@ -174,27 +175,20 @@ function scanHeaderRange(line: string, index = 0, end = trimEndIndex(line)): Ran if (start === null) return null; const afterFirst = scanRangeSeparator(line, start.nextIndex, end); - if (afterFirst !== null) { - const endNumber = scanLineNumber(line, afterFirst, end); - if (endNumber === null) return null; - return { - range: { start: { line: start.line }, end: { line: endNumber.line } }, - nextIndex: skipWhitespace(line, endNumber.nextIndex, end), - }; - } - // Shorthand: bare `A` treated as `A..A`. Trailing non-whitespace past - // `cursor` signals a malformed header (caller verifies). + if (afterFirst === null) return null; + const endNumber = scanLineNumber(line, afterFirst, end); + if (endNumber === null) return null; return { - range: { start: { line: start.line }, end: { line: start.line } }, - nextIndex: skipWhitespace(line, start.nextIndex, end), + range: { start: { line: start.line }, end: { line: endNumber.line } }, + nextIndex: skipWhitespace(line, endNumber.nextIndex, end), }; } /** - * Consume an optional range separator (whitespace, `-`, `..`, or `…`) - * after the first number in a header. Returns the index of the second - * number, or `null` when the next non-whitespace char isn't a digit - * (i.e. we're looking at a single-line shorthand). + * Consume the mandatory range separator (whitespace, `-`, `..`, or `…`) + * between the two numbers of a hunk-header range. Returns the index of + * the second number, or `null` when the separator is missing or no digit + * follows it. */ function scanRangeSeparator(line: string, index: number, end: number): number | null { let cursor = index; @@ -231,8 +225,9 @@ interface TargetScan { } /** - * Scan the anchor portion of a hunk header. Accepts `BOF`, `EOF`, `A B` - * (range), or `A` (single-line shorthand for `A A`). + * Scan the anchor portion of a hunk header. Accepts `BOF`, `EOF`, or + * `A B` (range). Single-number anchors are NOT accepted; callers must + * spell single-line ranges as `A A`. */ function scanHunkAnchor(line: string, start: number, end: number): TargetScan | null { const cursor = skipWhitespace(line, start, end); @@ -252,9 +247,8 @@ interface ParsedHunkHeader { } /** - * Parse a bare hunk-header line: `A B` (range), `A` (single-line shorthand - * for `A A`), or the keywords `BOF` / `EOF`. Returns `null` for lines that - * do not match the shape. + * Parse a bare hunk-header line: `A B` (range) or the keywords + * `BOF` / `EOF`. Returns `null` for lines that do not match the shape. */ function tryParseHunkHeader(line: string): ParsedHunkHeader | null { const end = trimEndIndex(line); @@ -366,10 +360,10 @@ function classifyLine(line: string, lineNum: number): Token { } } - // Hunk header lines start with a digit (range / single-line) or the - // keyword `BOF` / `EOF`. `@@`-bracketed forms are intentionally NOT - // accepted here — they fall through to `raw` and the parser rejects - // them as apply_patch contamination. + // Hunk header lines are `A B` (two numbers) or the keyword `BOF` / + // `EOF`. `@@`-bracketed forms are intentionally NOT accepted here — + // they fall through to `raw` and the parser rejects them as + // apply_patch contamination. const isHunkLead = isNonZeroDigitCode(firstCode) || line.startsWith(BOF_ANCHOR) || line.startsWith(EOF_ANCHOR); if (isHunkLead) { const hunk = tryParseHunkHeader(line); diff --git a/packages/hashline/test/leniency.test.ts b/packages/hashline/test/leniency.test.ts index bafe7b71e..db2c9906a 100644 --- a/packages/hashline/test/leniency.test.ts +++ b/packages/hashline/test/leniency.test.ts @@ -7,12 +7,12 @@ function applyPatch(text: string, diff: string): string { const FILE = "a\nb\nc\nd\ne"; -describe("hashline core — hunk header shorthand", () => { - it("accepts `A` as `A..A` (single-line shorthand)", () => { - expect(applyPatch(FILE, "2\n+B")).toBe("a\nB\nc\nd\ne"); +describe("hashline core — hunk header forms", () => { + it("rejects a bare single-number hunk header (single-line shorthand removed)", () => { + expect(() => parsePatch("2\n+B")).toThrow(/single-number hunk header/); }); - it("an empty `A..A` deletes the line", () => { + it("an empty `A A` deletes the line", () => { expect(applyPatch(FILE, "2 2")).toBe("a\nc\nd\ne"); });