diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 724ed5adb..58524dfd2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Removed the `DEL`, `DEL.BLK`, `COPY`, and `COPY.BLK` hashline edit operations. Use `CUT` / `CUT.BLK` for deletion; removed content remains available to `PASTE`. + ### Added - Added server-name autocomplete for `/mcp` commands (`enable`, `disable`, `test`, `remove`, `reconnect`, `reauth`, `unauth`) using configured and runtime-discovered MCP servers. diff --git a/packages/coding-agent/src/edit/hashline/execute.ts b/packages/coding-agent/src/edit/hashline/execute.ts index a000c8737..a2cfde969 100644 --- a/packages/coding-agent/src/edit/hashline/execute.ts +++ b/packages/coding-agent/src/edit/hashline/execute.ts @@ -210,7 +210,6 @@ function renderSection( }; } - export async function executeHashlineSingle( options: ExecuteHashlineSingleOptions, ): Promise> { diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index ea49810fb..ee81caf3c 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Removed `DEL`, `DEL.BLK`, `COPY`, and `COPY.BLK` from the patch language. Use `CUT` / `CUT.BLK` for deletion; a cut does not require a following `PASTE` and leaves the removed content available to later pastes. + ### Added - Added clipboard ops: `CUT N.=M` captures lines into a register (and deletes them), `CUT.BLK N` captures tree-sitter blocks, and `PASTE.PRE|POST N` / `PASTE.HEAD|TAIL` / `PASTE.BLK.POST N` insert the captured lines without retyping. The register flows top-to-bottom across sections, so content moves between files in one patch; `PASTE` does not consume it and the last capture wins. @@ -10,7 +14,7 @@ ### Changed -- Regrouped `grammar.lark` around shared shapes — one `target` rule (`N.=M` | `.BLK N`) for `DEL`/`CUT` and one `pos` rule (`PRE`/`POST`/`BLK.POST`/`HEAD`/`TAIL`) for `INS`/`PASTE` — collapsing per-op hunk rules. The accepted language is byte-identical. +- Simplified `grammar.lark` around shared target and position shapes, collapsing the concrete and block `CUT` forms plus the `INS` / `PASTE` position variants into their common grammar rules. ### Fixed diff --git a/packages/hashline/src/input.ts b/packages/hashline/src/input.ts index 00fa9312a..fc918f113 100644 --- a/packages/hashline/src/input.ts +++ b/packages/hashline/src/input.ts @@ -314,7 +314,6 @@ export class PatchSection { }); } - /** Anchor lines touched by this section, sorted ascending and deduplicated. */ collectAnchorLines(): readonly number[] { const lines = new Set(); diff --git a/packages/hashline/src/messages.ts b/packages/hashline/src/messages.ts index 71c6369f0..673a7d7fa 100644 --- a/packages/hashline/src/messages.ts +++ b/packages/hashline/src/messages.ts @@ -263,7 +263,6 @@ export function ambiguousCloserSpareMessage( export const UNRESOLVED_BLOCK_INTERNAL = "internal error: unresolved block edit reached the applier (resolveBlockEdits was not run)."; - /** `REM` received a body row or coexists with line edits. */ export const REM_TAKES_NO_BODY = "`REM` deletes the whole file and takes no body rows or line ops. Issue it alone under the header."; @@ -272,7 +271,6 @@ export const REM_TAKES_NO_BODY = export const MOVE_TAKES_NO_BODY = "`MV DEST` does not take body rows. Put line edits above the `MV` row; the destination path follows `MV` on the same line."; - /** `CUT N.=M` hunk received a body row. */ export const CUT_TAKES_NO_BODY = `\`CUT N${HL_RANGE_SEP}M\` captures + deletes lines and takes no body rows. To replace lines with new content, use \`SWAP N${HL_RANGE_SEP}M:\`.`; diff --git a/packages/hashline/test/clipboard.test.ts b/packages/hashline/test/clipboard.test.ts index 9e84d4e1f..7b755b6bf 100644 --- a/packages/hashline/test/clipboard.test.ts +++ b/packages/hashline/test/clipboard.test.ts @@ -178,7 +178,6 @@ describe("clipboard across sections and batches", () => { expect(fs.get("b.ts")).toBe("b1\nmove1\nmove2\n"); }); - it("applies a batch-local CUT without requiring PASTE", async () => { const a = "l1\nl2\n"; const { fs, patcher, tags } = taggedPatcher([["a.ts", a]]); diff --git a/packages/hashline/test/format-v2.test.ts b/packages/hashline/test/format-v2.test.ts index a2647daa2..007b68b40 100644 --- a/packages/hashline/test/format-v2.test.ts +++ b/packages/hashline/test/format-v2.test.ts @@ -52,8 +52,9 @@ describe("hashline format v4", () => { }); it("does not recognize removed DEL or COPY headers", () => { - expect(() => parsePatch("DEL 2")).toThrow(/payload line has no preceding hunk header/); - expect(() => parsePatch("COPY 2")).toThrow(/payload line has no preceding hunk header/); + for (const header of ["DEL 2", "DEL.BLK 2", "COPY 2", "COPY.BLK 2"]) { + expect(() => parsePatch(header)).toThrow(/payload line has no preceding hunk header/); + } }); it("auto-pipes bare body rows as literal text", () => { diff --git a/packages/metaharness/adapters/edit/prompts/benchmark-system.md b/packages/metaharness/adapters/edit/prompts/benchmark-system.md index 992e4487c..6881ea6b0 100644 --- a/packages/metaharness/adapters/edit/prompts/benchmark-system.md +++ b/packages/metaharness/adapters/edit/prompts/benchmark-system.md @@ -3,13 +3,12 @@ You are participating in a code-edit benchmark inside a repository with {{#if mu This benchmark is scored on exactness. Get the edit right. ## Important constraints -- Make the minimum change necessary. Do not refactor, improve, or clean up other code. -- If you see multiple similar patterns, only change the ONE that is buggy (there is only one intended mutation). -- Preserve exact code structure. Do not rearrange statements or change formatting. +- Make exactly the change the task specifies — nothing more. Do not refactor, improve, or clean up other code. +- Tasks range from single-token fixes to multi-hunk block rewrites. When the task shows replacement code, reproduce it byte-for-byte: indentation, tabs vs spaces, and blank lines included. +- If the file contains multiple similar regions, change only the one(s) the task identifies. - Your output is verified by exact text diff against an expected fixture. Equivalent code, reordered imports, reordered object keys, or formatting changes will fail. -- Prefer copying the original line(s) and changing only the specific token(s) required. Do not rewrite whole statements. - Never modify comments or license headers unless the task explicitly asks. -- Re-read the changed region after editing to confirm you only touched the intended line(s). +- Re-read the changed region after editing to confirm it matches the task exactly. {{#if multiFile}}- Only modify the file(s) referenced by the task or follow-up messages. Leave all other files unchanged. {{/if}} ## Process diff --git a/packages/typescript-edit-benchmark/fixtures.tar.gz b/packages/typescript-edit-benchmark/fixtures.tar.gz index 324f11bb4..0080f0d40 100644 Binary files a/packages/typescript-edit-benchmark/fixtures.tar.gz and b/packages/typescript-edit-benchmark/fixtures.tar.gz differ diff --git a/packages/typescript-edit-benchmark/package.json b/packages/typescript-edit-benchmark/package.json index 8ec81cceb..f286c6020 100644 --- a/packages/typescript-edit-benchmark/package.json +++ b/packages/typescript-edit-benchmark/package.json @@ -26,7 +26,8 @@ "test": "bun test --parallel", "fix": "biome check --write --unsafe .", "fmt": "biome format --write .", - "generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4" + "generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source", + "edit-shapes": "bun run src/edit-shape-stats.ts" }, "dependencies": { "@babel/generator": "catalog:", diff --git a/packages/typescript-edit-benchmark/src/edit-shape-stats.ts b/packages/typescript-edit-benchmark/src/edit-shape-stats.ts new file mode 100755 index 000000000..1b724deff --- /dev/null +++ b/packages/typescript-edit-benchmark/src/edit-shape-stats.ts @@ -0,0 +1,317 @@ +#!/usr/bin/env bun +/** + * Measure the empirical shape of real agent edits from session logs. + * + * Scans session JSONL files for successful `edit`/`apply_patch` tool calls and + * parses the diff in each tool result (both the numbered ` NNN|` format and + * unified `@@` diffs) into change runs — maximal runs of `-`/`+` rows with no + * context row between them — so unchanged context never counts as changed and + * every run counts as its own hunk. + * + * Shapes are reported at two levels: + * - per tool call — one sample per edit call; + * - per user request — all edit calls between one user message and the next + * grouped into one sample (one benchmark fixture ≙ one prompt, and the + * runner allows multiple edit calls per prompt). Retried/overlapping edits + * within a request are summed, so request sizes lean high, and a request + * spans every file a long agentic turn touched — an upper bound for + * single-file fixtures. + * + * Parse coverage is printed so a format skew cannot silently bias the numbers. + * Reference measurement (2026-07, 2,000 newest sessions against this repo, + * 99.9% coverage): + * + * per request: changed lines: 1 → 6% 2-5 → 11% 6-20 → 20% 21-60 → 25% 61+ → 39% + * hunks: 1 → 11% 2 → 8% 3+ → 81% (median 40 changed lines) + * per call: changed lines: 1 → 23% 2-5 → 30% 6-20 → 29% 21-60 → 14% 61+ → 5% + * hunks: 1 → 48% 2 → 21% 3+ → 31% (median 5 changed lines) + * op mix (both levels): replace 55% insert 32% delete 12% + * + * `generate.ts` fixtures are single-file, single-prompt tasks: the suite is + * calibrated to sit between the two levels — per-call shapes for token fixes, + * request-leaning shapes (large blocks, multi-hunk composites) for the rest. + */ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { parseArgs } from "node:util"; +import { computeDefaultSessionDir } from "@oh-my-pi/pi-coding-agent/session/session-paths"; +import { FileSessionStorage } from "@oh-my-pi/pi-coding-agent/session/session-storage"; + +const EDIT_TOOL_NAMES: Record = { edit: true, apply_patch: true }; +const SIZE_BUCKETS: Array<[string, number]> = [ + ["1", 1], + ["2-5", 5], + ["6-20", 20], + ["21-60", 60], + ["61+", Number.POSITIVE_INFINITY], +]; + +/** One maximal run of removed/added rows with no context row in between. */ +interface ChangeRun { + oldCount: number; + newCount: number; +} + +interface Args { + sessionsDir: string; + maxSessions: number; +} + +function parseArguments(): Args { + const { values } = parseArgs({ + options: { + "sessions-dir": { type: "string" }, + repo: { type: "string", default: path.resolve(import.meta.dir, "../../..") }, + "max-sessions": { type: "string", default: "2000" }, + }, + }); + const repo = path.resolve(values.repo ?? "."); + return { + sessionsDir: values["sessions-dir"] ?? computeDefaultSessionDir(repo, new FileSessionStorage()), + maxSessions: parseInt(values["max-sessions"] ?? "2000", 10), + }; +} + +const ROW_PIPE_RE = /^([ +-])\s*\d+\|/; +const ROW_SPACE_RE = /^([ +-])\s*\d+( |$)/; + +/** Collects change runs; a context row or elision closes the current run. */ +class RunCollector { + runs: ChangeRun[] = []; + #current: ChangeRun | null = null; + + mark(kind: string): void { + if (kind === " ") { + this.flush(); + return; + } + this.#current ??= { oldCount: 0, newCount: 0 }; + if (kind === "-") this.#current.oldCount++; + else this.#current.newCount++; + } + + flush(): void { + if (this.#current && (this.#current.oldCount > 0 || this.#current.newCount > 0)) { + this.runs.push(this.#current); + } + this.#current = null; + } + + finish(): ChangeRun[] | null { + this.flush(); + return this.runs.length > 0 ? this.runs : null; + } +} + +/** + * Parse the numbered diff format emitted in edit tool-result details + * (` NNN|context`, `-NNN|removed`, `+NNN|added`; older space-delimited rows). + * Elisions appear as blank rows or `...` rows. Returns null when any row is + * unparseable. + */ +function runsFromNumberedDiff(diff: string): ChangeRun[] | null { + const lines = diff.split("\n"); + const usesPipe = lines.slice(0, 10).some(line => ROW_PIPE_RE.test(line)); + const collector = new RunCollector(); + for (const line of lines) { + const row = ROW_PIPE_RE.exec(line) ?? (usesPipe ? null : ROW_SPACE_RE.exec(line)); + if (row) { + collector.mark(row[1]); + } else if (line.trim() === "" || line.trim() === "..." || line.trim() === "…" || line.trim().endsWith("|...")) { + collector.flush(); + } else { + return null; + } + } + return collector.finish(); +} + +/** + * Parse a unified diff (standard `@@ -a,b +c,d @@` hunks or anchor-style `@@` + * headers): `-` old, `+` new, leading space or blank = context. + */ +function runsFromUnifiedDiff(diff: string): ChangeRun[] | null { + const collector = new RunCollector(); + for (const line of diff.split("\n")) { + if (line.startsWith("@@")) { + collector.flush(); + } else if (line.startsWith("--- ") || line.startsWith("+++ ") || line.startsWith("Index:")) { + // File headers. + } else if (line.startsWith("-") || line.startsWith("+")) { + collector.mark(line[0]); + } else if (line.startsWith(" ") || line === "") { + collector.mark(" "); + } else { + return null; + } + } + return collector.finish(); +} + +/** Parse a tool-result diff in whichever format it uses. */ +function runsFromDiff(diff: string): { runs: ChangeRun[] | null; format: "numbered" | "unified" } { + const head = diff.trimStart(); + if (head.startsWith("@@") || head.startsWith("--- ")) { + return { runs: runsFromUnifiedDiff(diff), format: "unified" }; + } + return { runs: runsFromNumberedDiff(diff), format: "numbered" }; +} + +/** Per-session parse results: change runs per edit call and per user request, plus coverage counters. */ +interface SessionScan { + edits: ChangeRun[][]; + /** All edit-call runs between one user message and the next, concatenated. */ + requests: ChangeRun[][]; + formats: Map; + skipped: number; +} + +/** Extract change runs for every successful edit in one session JSONL file. */ +async function collectSessionEdits(sessionFile: string): Promise { + const scan: SessionScan = { edits: [], requests: [], formats: new Map(), skipped: 0 }; + const stream = Bun.file(sessionFile).stream(); + const decoder = new TextDecoder(); + let buffer = ""; + let currentRequest: ChangeRun[] = []; + + const flushRequest = (): void => { + if (currentRequest.length > 0) scan.requests.push(currentRequest); + currentRequest = []; + }; + + const handleRecord = (record: unknown): void => { + if (!record || typeof record !== "object") return; + const rec = record as Record; + if (rec.type !== "message" || !rec.message || typeof rec.message !== "object") return; + const message = rec.message as Record; + // A user message starts a new request; everything until the next one + // belongs to the same prompt (one benchmark fixture ≙ one request). + if (message.role === "user") { + flushRequest(); + return; + } + if (message.role !== "toolResult" || !EDIT_TOOL_NAMES[String(message.toolName)] || message.isError) return; + const details = + message.details && typeof message.details === "object" + ? (message.details as Record) + : undefined; + const diff = details && typeof details.diff === "string" ? details.diff : null; + const op = details && typeof details.op === "string" ? details.op : null; + if (!diff || (op && op !== "update")) return; + const { runs, format } = runsFromDiff(diff); + if (runs) { + scan.edits.push(runs); + currentRequest.push(...runs); + scan.formats.set(format, (scan.formats.get(format) ?? 0) + 1); + } else { + scan.skipped++; + } + }; + + for await (const chunk of stream) { + buffer += decoder.decode(chunk, { stream: true }); + const parsed = Bun.JSONL.parseChunk(buffer); + for (const value of parsed.values) handleRecord(value); + buffer = buffer.slice(parsed.read); + } + for (const value of Bun.JSONL.parse(`${buffer}\n`)) handleRecord(value); + flushRequest(); + return scan; +} + +function sizeBucket(changedLines: number): string { + for (const [label, max] of SIZE_BUCKETS) { + if (changedLines <= max) return label; + } + return SIZE_BUCKETS[SIZE_BUCKETS.length - 1][0]; +} + +function printDistribution(title: string, counts: Map, order?: string[]): void { + const total = [...counts.values()].reduce((a, b) => a + b, 0); + const keys = order ?? [...counts.keys()].sort(); + console.log(title); + for (const key of keys) { + const count = counts.get(key) ?? 0; + console.log(` ${key.padEnd(6)} ${String(count).padStart(6)} ${((count / total) * 100).toFixed(1)}%`); + } +} + +async function main(): Promise { + const args = parseArguments(); + const sessionFiles = (await fs.readdir(args.sessionsDir)) + .filter(name => name.endsWith(".jsonl")) + .sort() + .reverse() + .slice(0, args.maxSessions > 0 ? args.maxSessions : undefined) + .map(name => path.join(args.sessionsDir, name)); + console.log(`Scanning ${sessionFiles.length} session files in ${args.sessionsDir}`); + + interface ShapeCounters { + sizes: Map; + hunks: Map; + ops: Map; + changed: number[]; + } + const newCounters = (): ShapeCounters => ({ sizes: new Map(), hunks: new Map(), ops: new Map(), changed: [] }); + const accumulate = (counters: ShapeCounters, runs: ChangeRun[]): void => { + const changed = runs.reduce((sum, run) => sum + Math.max(run.oldCount, run.newCount), 0); + counters.changed.push(changed); + counters.sizes.set(sizeBucket(changed), (counters.sizes.get(sizeBucket(changed)) ?? 0) + 1); + const hunkKey = runs.length >= 3 ? "3+" : String(runs.length); + counters.hunks.set(hunkKey, (counters.hunks.get(hunkKey) ?? 0) + 1); + for (const run of runs) { + const op = run.oldCount === 0 ? "insert" : run.newCount === 0 ? "delete" : "replace"; + counters.ops.set(op, (counters.ops.get(op) ?? 0) + 1); + } + }; + const report = (label: string, counters: ShapeCounters): void => { + counters.changed.sort((a, b) => a - b); + const median = counters.changed[Math.floor(counters.changed.length / 2)] ?? 0; + console.log(`\n=== ${label}: ${counters.changed.length} samples, median changed lines: ${median}`); + printDistribution( + "changed lines:", + counters.sizes, + SIZE_BUCKETS.map(([bucketLabel]) => bucketLabel), + ); + printDistribution("hunks:", counters.hunks, ["1", "2", "3+"]); + printDistribution("op mix:", counters.ops, ["replace", "insert", "delete"]); + }; + + const perCall = newCounters(); + const perRequest = newCounters(); + const formatCounts = new Map(); + let skipped = 0; + + let next = 0; + const workers = Array.from({ length: 16 }, async () => { + while (next < sessionFiles.length) { + const file = sessionFiles[next++]; + const scan = await collectSessionEdits(file).catch( + (): SessionScan => ({ edits: [], requests: [], formats: new Map(), skipped: 0 }), + ); + skipped += scan.skipped; + for (const [format, count] of scan.formats) { + formatCounts.set(format, (formatCounts.get(format) ?? 0) + count); + } + for (const runs of scan.edits) accumulate(perCall, runs); + for (const runs of scan.requests) accumulate(perRequest, runs); + } + }); + await Promise.all(workers); + + if (perCall.changed.length === 0) { + console.error("No parseable edits found."); + return 1; + } + const parsedTotal = [...formatCounts.values()].reduce((a, b) => a + b, 0); + const coverage = parsedTotal + skipped > 0 ? parsedTotal / (parsedTotal + skipped) : 0; + const formatSummary = [...formatCounts.entries()].map(([format, count]) => `${format}=${count}`).join(", "); + console.log( + `\nParse coverage: ${parsedTotal}/${parsedTotal + skipped} diff results (${(coverage * 100).toFixed(1)}%; ${formatSummary})`, + ); + report("per user request (calibration target)", perRequest); + report("per tool call", perCall); + return 0; +} + +process.exit(await main()); diff --git a/packages/typescript-edit-benchmark/src/generate.ts b/packages/typescript-edit-benchmark/src/generate.ts index 96ddf71d7..d970ac0a5 100644 --- a/packages/typescript-edit-benchmark/src/generate.ts +++ b/packages/typescript-edit-benchmark/src/generate.ts @@ -12,20 +12,38 @@ * - Dense code: Minimal whitespace makes context harder to read * - Deep nesting: Whitespace-sensitive edits at high indent levels * - * Difficulty modes control both FILE SELECTION and PROMPT DETAIL: - * - easy: Short files, unique lines, line number given - * - medium: Medium files, function context given + * Difficulty modes control both FILE SELECTION and PROMPT DETAIL. Every prompt + * fully determines the byte-exact fix (before/after blocks, validated by + * re-solving them against the expected output); tiers vary how much locating + * the model must do: + * - easy: Short files, unique lines, exact line number given + * - medium: Medium files, containing function given * - hard: Long files with similar blocks, no location hint - * - nightmare: Long files where target line repeats, minimal info + * - nightmare: Long files where nearly identical regions repeat, no location hint */ import * as fs from "node:fs"; import * as path from "node:path"; import { parseArgs } from "node:util"; -import { TempDir } from "@oh-my-pi/pi-utils"; +import { prompt, TempDir } from "@oh-my-pi/pi-utils"; import { $ } from "bun"; import { diffLines } from "diff"; import { formatContent } from "./formatter"; +import { + commonPrefixLength, + commonSuffixLength, + LANGUAGE_BY_EXTENSION, + mergeNearbyPlacements, + type Placement, + pickFence, + placementsFromDiff, + type RenderedHunk, + renderHunks, + solveRenderedHunks, +} from "./hunks"; import { ALL_MUTATIONS, CATEGORY_MAP, type Mutation, type MutationInfo } from "./mutations"; +import identifierTaskTemplate from "./prompts/identifier-task.md" with { type: "text" }; +import mutationTaskTemplate from "./prompts/mutation-task.md" with { type: "text" }; +import structuralTaskTemplate from "./prompts/structural-task.md" with { type: "text" }; const SUPPORTED_EXTENSIONS = new Set([".js", ".jsx", ".ts", ".tsx"]); using DEFAULT_SOURCE_REPO_DIR = TempDir.createSync("@pi-mono-source"); @@ -64,6 +82,7 @@ interface CaseResult { filePath: string; formattedMutatedContent: string; formattedOriginalContent: string; + prompt: string; difficulty: Difficulty; difficultyScore: number; } @@ -71,7 +90,7 @@ interface CaseResult { interface Args { typescriptDir: string; output: string; - countPerType: number; + countScale: number; seed: number; categories: string | null; difficulty: string; @@ -84,7 +103,7 @@ function parseArguments(): Args { options: { "typescript-dir": { type: "string", default: DEFAULT_SOURCE_REPO_DIR.path() }, output: { type: "string", default: DEFAULT_OUTPUT }, - "count-per-type": { type: "string", default: "20" }, + "count-scale": { type: "string", default: "1" }, seed: { type: "string", default: "42" }, categories: { type: "string" }, difficulty: { type: "string", default: "easy,medium,hard,nightmare" }, @@ -96,7 +115,7 @@ function parseArguments(): Args { return { typescriptDir: values["typescript-dir"] ?? DEFAULT_SOURCE_REPO_DIR.path(), output: values.output ?? DEFAULT_OUTPUT, - countPerType: parseInt(values["count-per-type"] ?? "20", 10), + countScale: Number.parseFloat(values["count-scale"] ?? "1"), seed: parseInt(values.seed ?? "42", 10), categories: values.categories ?? null, difficulty: values.difficulty ?? "easy,medium,hard,nightmare", @@ -105,6 +124,59 @@ function parseArguments(): Args { }; } +/** Target changed-line range for a generated case. */ +type SizeRange = [number, number]; + +/** + * Per-mutation case counts and changed-line targets, calibrated against the + * empirical edit-shape distribution measured from real agent sessions + * (`edit-shape-stats.ts`): 1 → 27%, 2-5 → 30%, 6-20 → 26%, 21-60 → 13%, + * 61+ → 4%. Token-level mutations supply the 1-line mass; small structural and + * multi-site mutations the 2-5 band; block-level mutations cycle through the + * larger bands via `sizes`. + */ +interface MutationPlan { + count: number; + sizes?: SizeRange[]; +} + +const BLOCK_SIZES: SizeRange[] = [ + [6, 20], + [6, 20], + [21, 60], + [6, 20], + [21, 60], + [6, 20], + [61, 150], + [21, 60], +]; + +const MUTATION_PLANS: Record = { + "swap-comparison": { count: 2 }, + "swap-equality": { count: 2 }, + "swap-logical": { count: 2 }, + "remove-negation": { count: 2 }, + "swap-increment-decrement": { count: 2 }, + "swap-arithmetic": { count: 2 }, + "flip-boolean": { count: 2 }, + "remove-optional-chain": { count: 2 }, + "swap-call-args": { count: 2 }, + "swap-nullish": { count: 2 }, + "swap-regex-quantifier": { count: 2 }, + "unicode-hyphen": { count: 2 }, + "off-by-one": { count: 2 }, + "swap-adjacent-lines": { count: 6 }, + "duplicate-line-flip": { count: 6 }, + "identifier-multi-edit": { count: 8 }, + "swap-if-else": { count: 6 }, + "wrap-redundant-if": { count: 12, sizes: BLOCK_SIZES }, + "swap-sibling-blocks": { count: 12, sizes: BLOCK_SIZES }, + "duplicate-block": { count: 6, sizes: BLOCK_SIZES }, + "move-distant-block": { count: 14, sizes: BLOCK_SIZES }, + "remove-case-label": { count: 8 }, + "composite-multi-edit": { count: 14 }, +}; + async function ensureSourceRepo(typescriptDir: string): Promise { if (fs.existsSync(typescriptDir)) { const packagesDir = path.join(typescriptDir, "packages"); @@ -378,69 +450,295 @@ function recordRegion(usedLines: Map, filePath: string, lineNu usedLines.set(filePath, used); } +function escapeRegExp(value: string): string { + return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +} + +/** Where a bug sits for medium prompts: containing function, else file-third region. */ +function locateBug( + content: string, + lineNumber: number, + lineCount: number, +): { functionName?: string; region?: "top" | "middle" | "end" } { + for (const [name, start, end] of findFunctionRanges(content)) { + if (lineNumber >= start && lineNumber <= end) return { functionName: name }; + } + const ratio = lineCount > 0 ? lineNumber / lineCount : 0; + return { region: ratio < 0.33 ? "top" : ratio < 0.66 ? "middle" : "end" }; +} + +/** + * Rename-style prompt for identifier-multi-edit: names the misspelled and + * correct identifiers so "replace every occurrence" admits exactly one + * byte-exact answer. Returns null when a plain word-boundary rename does not + * reproduce the expected file (e.g. the misspelling collides with other text). + */ +function buildIdentifierPrompt( + filePath: string, + info: MutationInfo, + difficulty: Difficulty, + input: string, + expected: string, +): string | null { + const pair = info.identifier; + if (!pair) return null; + const wordRe = new RegExp(`(? { + if (wordRe.test(line)) affectedLines.push(index + 1); + wordRe.lastIndex = 0; + }); + if (affectedLines.length === 0) return null; + + return prompt.render(identifierTaskTemplate, { + filename: path.basename(filePath), + correct: pair.correct, + misspelled: pair.misspelled, + count: affectedLines.length, + affectedLines: difficulty === "easy" ? affectedLines : undefined, + }); +} + +/** Mutations whose fixtures read as natural instructions instead of raw before/after patches. */ +type StructuralKind = + | "case-label" + | "duplicate-block" + | "move-block" + | "wrap-if" + | "swap-blocks" + | "swap-lines" + | "swap-if-else"; + +const STRUCTURAL_KINDS: Record = { + "remove-case-label": "case-label", + "duplicate-block": "duplicate-block", + "move-distant-block": "move-block", + "wrap-redundant-if": "wrap-if", + "swap-sibling-blocks": "swap-blocks", + "swap-adjacent-lines": "swap-lines", + "swap-if-else": "swap-if-else", +}; + +function countTrimmedLine(lines: string[], needle: string): number { + let count = 0; + for (const line of lines) { + if (line.trim() === needle) count++; + } + return count; +} + +/** Whether a line is a usable prose anchor (contains an identifier, not just punctuation like `);`). */ +function isAnchorLine(line: string): boolean { + return /[A-Za-z_$]{2,}/.test(line); +} + +function firstNonEmptyTrimmed(lines: string[]): string { + for (const line of lines) { + const trimmed = line.trim(); + if (trimmed) return trimmed; + } + return ""; +} + +/** A placement reduced to its changed core (shared context trimmed away). */ +function trimPlacement( + inputLines: string[], + placement: Placement, +): { start: number; oldCore: string[]; newCore: string[] } { + const old = inputLines.slice(placement.start, placement.start + placement.oldLen); + const fresh = placement.newLines; + const prefix = commonPrefixLength(old, fresh); + const suffix = commonSuffixLength(old, fresh, Math.min(old.length, fresh.length) - prefix); + return { + start: placement.start + prefix, + oldCore: old.slice(prefix, old.length - suffix), + newCore: fresh.slice(prefix, fresh.length - suffix), + }; +} + +/** Nearest non-blank line trimmed, scanning from `index` in `direction`. */ +function nearestVisibleLine(lines: string[], index: number, direction: 1 | -1): string { + for (let i = index; i >= 0 && i < lines.length; i += direction) { + const trimmed = lines[i].trim(); + if (trimmed) return trimmed; + } + return ""; +} + +/** + * Derive instruction data for a structural mutation from its placements. + * Every referenced anchor line is validated to occur exactly the expected + * number of times, so the instruction identifies a unique edit; returns null + * when the instruction would be ambiguous (the caller then rejects the + * candidate and retries — structural kinds never ship raw-patch fallbacks). + */ +function structuralPromptData( + kind: StructuralKind, + inputLines: string[], + placements: Placement[], +): Record | null { + const cores = placements.map(placement => trimPlacement(inputLines, placement)); + const visible = (lines: string[]): string[] => lines.filter(line => line.trim() !== ""); + + switch (kind) { + case "case-label": { + if (cores.length !== 1) return null; + const { start, oldCore, newCore } = cores[0]; + const added = visible(newCore); + if (visible(oldCore).length !== 0 || added.length !== 1) return null; + const label = added[0].trim(); + const before = nearestVisibleLine(inputLines, start + oldCore.length, 1); + if (!label.startsWith("case") || !(before.startsWith("case") || before.startsWith("default"))) return null; + if (countTrimmedLine(inputLines, before) !== 1) return null; + return { label, before }; + } + case "duplicate-block": { + if (cores.length !== 1) return null; + const { oldCore, newCore } = cores[0]; + if (visible(newCore).length !== 0 || visible(oldCore).length === 0) return null; + const head = firstNonEmptyTrimmed(oldCore); + if (!head || countTrimmedLine(inputLines, head) !== 2) return null; + return { head }; + } + case "move-block": { + if (cores.length !== 2) return null; + const insert = cores.find(core => visible(core.oldCore).length === 0); + const remove = cores.find(core => visible(core.newCore).length === 0); + if (!insert || !remove || insert === remove) return null; + if (visible(remove.oldCore).join("\n") !== visible(insert.newCore).join("\n")) return null; + const head = firstNonEmptyTrimmed(remove.oldCore); + const destination = nearestVisibleLine(inputLines, insert.start + insert.oldCore.length, 1); + const currentPrev = nearestVisibleLine(inputLines, remove.start - 1, -1); + if (![head, destination, currentPrev].every(isAnchorLine)) return null; + for (const anchor of [head, destination, currentPrev]) { + if (countTrimmedLine(inputLines, anchor) !== 1) return null; + } + return { head, destination, currentPrev }; + } + case "wrap-if": { + if (cores.length !== 1) return null; + const { start, oldCore } = cores[0]; + const relative = oldCore.findIndex(line => line.trim() === "if (true) {"); + if (relative < 0) return null; + return { wrapperLine: start + relative + 1 }; + } + case "swap-blocks": + case "swap-lines": { + const merged = mergeNearbyPlacements(inputLines, placements, 2); + let firstHead = ""; + let secondHead = ""; + if (merged.length === 1) { + const { oldCore, newCore } = trimPlacement(inputLines, merged[0]); + firstHead = firstNonEmptyTrimmed(oldCore); + secondHead = firstNonEmptyTrimmed(newCore); + } else if (cores.length === 2) { + // jsdiff renders `A,B -> B,A` as insert-A-before-B plus delete-A-after-B, + // with all of B as unchanged context in between; match the moved body. + const insert = cores.find(core => visible(core.oldCore).length === 0); + const remove = cores.find(core => visible(core.newCore).length === 0); + if (!insert || !remove || insert === remove) return null; + if (visible(remove.oldCore).join("\n") !== visible(insert.newCore).join("\n")) return null; + secondHead = firstNonEmptyTrimmed(insert.newCore); + firstHead = nearestVisibleLine(inputLines, insert.start + insert.oldCore.length, 1); + } else { + return null; + } + if (!firstHead || !secondHead || firstHead === secondHead) return null; + if (!isAnchorLine(firstHead) || !isAnchorLine(secondHead)) return null; + if (countTrimmedLine(inputLines, firstHead) !== 1 || countTrimmedLine(inputLines, secondHead) !== 1) + return null; + return { firstHead, secondHead }; + } + case "swap-if-else": { + if (cores.length > 2) return null; + const condition = nearestVisibleLine(inputLines, cores[0].start - 1, -1); + if (!condition.startsWith("if") && !condition.includes(" if ")) return null; + if (countTrimmedLine(inputLines, condition) !== 1) return null; + return { condition }; + } + } +} + +/** + * Build the task prompt for a mutation case, or null when no uniquely solvable + * prompt exists (the caller then retries with a different candidate). + * + * Structural mutations render as natural instructions ("move X back before Y") + * with the expected after-state as reference; identifier renames as a + * replace-every-occurrence instruction; everything else as explicit + * before/after blocks. All variants are validated by re-solving the hunks + * against the expected output, so every prompt admits exactly one byte-exact + * answer. Difficulty only controls location detail. + */ function buildPrompt( filePath: string, mutation: Mutation, info: MutationInfo, difficulty: Difficulty, - entry: FileEntry, -): string { - const header = `# Fix the bug in \`${path.basename(filePath)}\``; - - const isStructural = mutation.category === "structural"; - const isMultiEdit = mutation.name === "identifier-multi-edit"; - - if (difficulty === "easy") { - const detail = mutation.describe(info); - const location = isStructural - ? `The issue starts around line ${info.lineNumber}.` - : `The issue is on line ${info.lineNumber}.`; - return [header, detail, location, mutation.fixHint].join("\n\n"); + input: string, + expected: string, +): string | null { + if (mutation.name === "identifier-multi-edit") { + return buildIdentifierPrompt(filePath, info, difficulty, input, expected); } - if (difficulty === "medium") { - const detail = mutation.describe(info); - const funcName = findContainingFunction(entry, info.lineNumber); - let location: string; - if (funcName) { - location = `The issue is in the \`${funcName}\` function.`; - } else { - const ratio = entry.lineCount ? info.lineNumber / entry.lineCount : 0; - if (ratio < 0.33) location = "The issue is near the top of the file."; - else if (ratio < 0.66) location = "The issue is around the middle of the file."; - else location = "The issue is near the end of the file."; - } - if (isMultiEdit) { - location += " The same error appears in multiple places."; - } - return [header, detail, location, mutation.fixHint].join("\n\n"); + const inputLines = input.replace(/\n$/, "").split("\n"); + const placements = placementsFromDiff(input, expected); + if (placements.length === 0 || placements.length > 5) return null; + const hunks = renderHunks(inputLines, placements); + if (!hunks) return null; + const solved = solveRenderedHunks(inputLines, hunks); + if (!solved || ensureTrailingNewline(solved.join("\n")) !== expected) return null; + + const language = LANGUAGE_BY_EXTENSION[path.extname(filePath).toLowerCase()] ?? ""; + const fence = pickFence(hunks); + const hunkContext = (hunk: RenderedHunk): { startLine: number | undefined } => ({ + startLine: !hunk.unique || difficulty === "easy" ? hunk.startLine : undefined, + }); + + const structuralKind = STRUCTURAL_KINDS[mutation.name]; + if (structuralKind) { + // Structural tasks read as natural instructions; a candidate whose + // instruction would be ambiguous is rejected so the caller retries — + // never downgraded to a raw before/after patch. + const data = structuralPromptData(structuralKind, inputLines, placements); + if (!data) return null; + const rendered = prompt.render(structuralTaskTemplate, { + filename: path.basename(filePath), + kind: structuralKind, + ...data, + fence, + language, + hunkCount: hunks.length, + hunks: hunks.map(hunk => ({ + // After-state regions are only interpretable with a position. + startLine: hunk.startLine, + newCode: hunk.newBlock.join("\n"), + })), + }); + return rendered.length <= 20_000 ? rendered : null; } - if (difficulty === "hard") { - const detail = mutation.describe(info); - if (isMultiEdit) { - return [header, detail, "Find and fix all occurrences of this issue."].join("\n\n"); - } - if (isStructural) { - return [header, detail, "The fix may involve multiple lines."].join("\n\n"); - } - return [header, detail, "Find and fix this issue."].join("\n\n"); - } - - // nightmare - if (isStructural) { - return [header, "There is a structural bug in this file.", "Track it down and fix it with a minimal edit."].join( - "\n\n", - ); - } - if (isMultiEdit) { - return [ - header, - "An identifier is consistently misspelled throughout this file.", - "Find all occurrences and fix them.", - ].join("\n\n"); - } - return [header, "There is a subtle bug in this file.", "Track it down and fix it with a minimal edit."].join("\n\n"); + const location = difficulty === "medium" ? locateBug(input, hunks[0].startLine, inputLines.length) : {}; + const rendered = prompt.render(mutationTaskTemplate, { + filename: path.basename(filePath), + name: mutation.name, + functionName: location.functionName, + region: location.region, + nightmare: difficulty === "nightmare", + fence, + language, + hunkCount: hunks.length, + hunks: hunks.map(hunk => ({ + ...hunkContext(hunk), + isDelete: hunk.newBlock.length === 0, + oldCode: hunk.oldBlock.join("\n"), + newCode: hunk.newBlock.join("\n"), + })), + }); + return rendered.length <= 20_000 ? rendered : null; } function createSeededRng(seed: number): () => number { @@ -625,6 +923,7 @@ function resolveFormattedMutationInfo( lineNumber, originalSnippet: snippetFromChangedLines(chosen.removedLines, originalLines, Math.max(1, chosen.oldStart)), mutatedSnippet: snippetFromChangedLines(chosen.addedLines, mutatedLines, lineNumber), + identifier: fallbackInfo.identifier, }; if (resolved.originalSnippet && !originalContent.includes(resolved.originalSnippet)) return null; @@ -640,6 +939,7 @@ async function generateCase( usedLines: Map, difficulty: Difficulty, minScore: number | null, + sizeRange: SizeRange | null, attemptLimit = 100, ): Promise { let candidates = getCandidatesForDifficulty(files, difficulty); @@ -676,12 +976,15 @@ async function generateCase( const positionalHunks = countPositionalLineHunks(entry.content, mutatedContent); const positionalChanges = countPositionalLineChanges(entry.content, mutatedContent); const isStructural = mutation.category === "structural"; - const isMultiEdit = mutation.name === "identifier-multi-edit"; + const isMultiEdit = mutation.multiHunk === true; if (!isStructural && !isMultiEdit) { if (changedHunks !== 1) continue; if (positionalHunks !== 1) continue; if (changedLines > 30 || positionalChanges > 30) continue; } + // Coarse pre-format lower bound: removed+added always >= the run-based + // metric, so anything below the minimum can be rejected before Prettier. + if (sizeRange && changedLines < sizeRange[0]) continue; if (!regionAvailable(usedLines, entry.path, rawInfo.lineNumber)) continue; @@ -707,6 +1010,26 @@ async function generateCase( const finalInfo = resolveFormattedMutationInfo(normalizedOriginal, normalizedMutated, rawInfo); if (!finalInfo) continue; + // Enforce the size target with the same metric edit-shape-stats.ts uses: + // sum of max(old, new) per change run, on the formatted input/expected. + if (sizeRange) { + const finalChanged = placementsFromDiff(normalizedMutated, normalizedOriginal).reduce( + (sum, placement) => sum + Math.max(placement.oldLen, placement.newLines.length), + 0, + ); + if (finalChanged < sizeRange[0] || finalChanged > sizeRange[1]) continue; + } + + const taskPrompt = buildPrompt( + entry.path, + mutation, + finalInfo, + difficulty, + normalizedMutated, + normalizedOriginal, + ); + if (!taskPrompt) continue; + recordRegion(usedLines, entry.path, rawInfo.lineNumber); return { caseId: "", @@ -715,6 +1038,7 @@ async function generateCase( filePath: entry.path, formattedMutatedContent: normalizedMutated, formattedOriginalContent: normalizedOriginal, + prompt: taskPrompt, difficulty, difficultyScore: diffScore, }; @@ -760,7 +1084,6 @@ async function buildCaseEntries(result: CaseResult, typescriptDir: string): Prom }; const isRepeated = entry.repeatedLines.has(lineContent); - const prompt = buildPrompt(result.filePath, result.mutation, result.finalInfo, result.difficulty, entry); const metadata = { mutation_type: result.mutation.name, @@ -796,7 +1119,7 @@ async function buildCaseEntries(result: CaseResult, typescriptDir: string): Prom return [ { name: `${caseDir}/input/${filename}`, content: result.formattedMutatedContent }, { name: `${caseDir}/expected/${filename}`, content: result.formattedOriginalContent }, - { name: `${caseDir}/prompt.md`, content: prompt }, + { name: `${caseDir}/prompt.md`, content: result.prompt }, { name: `${caseDir}/metadata.json`, content: JSON.stringify(metadata, null, 2) }, ]; } @@ -856,17 +1179,20 @@ async function main(): Promise { const fallbackOrder: Difficulty[] = ["hard", "medium", "easy"]; for (const mutation of mutations) { - const difficultiesForType = chooseDifficulties(difficulties, args.countPerType); + const plan = MUTATION_PLANS[mutation.name] ?? { count: 4 }; + const count = Math.max(1, Math.round(plan.count * args.countScale)); + const difficultiesForType = chooseDifficulties(difficulties, count); let generated = 0; - for (let index = 0; index < args.countPerType; index++) { + for (let index = 0; index < count; index++) { const difficulty = difficultiesForType[index]; - let result = await generateCase(rng, mutation, files, usedLines, difficulty, args.minScore); + const sizeRange = plan.sizes ? plan.sizes[index % plan.sizes.length] : null; + let result = await generateCase(rng, mutation, files, usedLines, difficulty, args.minScore, sizeRange); if (!result) { for (const fallback of fallbackOrder) { if (fallback === difficulty) continue; - result = await generateCase(rng, mutation, files, usedLines, fallback, 0); + result = await generateCase(rng, mutation, files, usedLines, fallback, 0, sizeRange); if (result) { console.log(`Note: ${mutation.name} case ${index + 1} fell back from ${difficulty} to ${fallback}`); break; @@ -889,8 +1215,8 @@ async function main(): Promise { if (generated === 0) { console.log(`Warning: No cases generated for ${mutation.name} (mutation may be too rare)`); - } else if (generated < args.countPerType) { - console.log(`Note: Only ${generated}/${args.countPerType} cases generated for ${mutation.name}`); + } else if (generated < count) { + console.log(`Note: Only ${generated}/${count} cases generated for ${mutation.name}`); } } diff --git a/packages/typescript-edit-benchmark/src/hunks.ts b/packages/typescript-edit-benchmark/src/hunks.ts new file mode 100644 index 000000000..bc04c15af --- /dev/null +++ b/packages/typescript-edit-benchmark/src/hunks.ts @@ -0,0 +1,264 @@ +/** + * Shared hunk machinery for fixture generation. + * + * `generate.ts` expresses every task as "replace these exact blocks in the + * input file". This module locates changed blocks, prepares prompt-visible + * before/after blocks whose old side occurs exactly once in the input + * (re-padding with context, or falling back to an explicit line number), and + * re-solves prepared hunks to prove a prompt admits exactly one byte-exact + * answer. Prompt prose itself lives in the Handlebars templates under + * `src/prompts/`; this module only supplies the block data. + */ +import { diffLines } from "diff"; + +/** Replace `oldLen` lines at `start` (0-based index into the input) with `newLines`. */ +export interface Placement { + start: number; + oldLen: number; + newLines: string[]; +} + +/** A prompt-visible change: replace (or delete) `oldBlock`, verbatim, with `newBlock`. */ +export interface RenderedHunk { + oldBlock: string[]; + newBlock: string[]; + /** 1-based line where `oldBlock` starts in the input file. */ + startLine: number; + /** Whether `oldBlock` occurs exactly once in the input; when false, prompts must state `startLine`. */ + unique: boolean; +} + +const MAX_UNIQUENESS_EXTENSION = 10; + +export function findBlockOccurrences(lines: string[], block: string[]): number[] { + if (block.length === 0 || block.length > lines.length) return []; + const hits: number[] = []; + const first = block[0]; + outer: for (let i = 0; i <= lines.length - block.length; i++) { + if (lines[i] !== first) continue; + for (let j = 1; j < block.length; j++) { + if (lines[i + j] !== block[j]) continue outer; + } + hits.push(i); + } + return hits; +} + +export function applyPlacements(lines: string[], placements: Placement[]): string[] { + const out: string[] = []; + let cursor = 0; + for (const placement of placements) { + out.push(...lines.slice(cursor, placement.start), ...placement.newLines); + cursor = placement.start + placement.oldLen; + } + out.push(...lines.slice(cursor)); + return out; +} + +export function commonPrefixLength(a: string[], b: string[]): number { + let n = 0; + while (n < a.length && n < b.length && a[n] === b[n]) n++; + return n; +} + +export function commonSuffixLength(a: string[], b: string[], maxLength: number): number { + let n = 0; + while (n < maxLength && a[a.length - 1 - n] === b[b.length - 1 - n]) n++; + return n; +} + +function splitDiffValue(value: string): string[] { + const lines = value.split("\n"); + if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop(); + return lines; +} + +/** + * Compute placements that transform `inputText` into `expectedText` via a + * line diff. Adjacent removed/added runs merge into one placement. + */ +export function placementsFromDiff(inputText: string, expectedText: string): Placement[] { + const placements: Placement[] = []; + let inputLine = 0; + let pendingStart = -1; + let pendingOldLen = 0; + let pendingNewLines: string[] = []; + + const flush = (): void => { + if (pendingStart < 0) return; + placements.push({ start: pendingStart, oldLen: pendingOldLen, newLines: pendingNewLines }); + pendingStart = -1; + pendingOldLen = 0; + pendingNewLines = []; + }; + + for (const change of diffLines(inputText, expectedText)) { + const lines = splitDiffValue(change.value); + if (!change.added && !change.removed) { + flush(); + inputLine += lines.length; + continue; + } + if (pendingStart < 0) pendingStart = inputLine; + if (change.removed) { + pendingOldLen += lines.length; + inputLine += lines.length; + } else { + pendingNewLines.push(...lines); + } + } + flush(); + return placements; +} + +/** + * Merge placements separated by at most `maxGap` context lines into one, so a + * moved statement renders as a single natural replace instead of a delete plus + * a re-insert whose blocks end on invisible blank lines. + */ +export function mergeNearbyPlacements(inputLines: string[], placements: Placement[], maxGap: number): Placement[] { + const merged: Placement[] = []; + for (const placement of placements) { + const previous = merged[merged.length - 1]; + const gap = previous ? placement.start - (previous.start + previous.oldLen) : -1; + if (previous && gap >= 0 && gap <= maxGap) { + const bridge = inputLines.slice(previous.start + previous.oldLen, placement.start); + previous.oldLen += gap + placement.oldLen; + previous.newLines = [...previous.newLines, ...bridge, ...placement.newLines]; + continue; + } + merged.push({ ...placement }); + } + return merged; +} + +/** + * Turn placements into prompt-visible hunks. Each hunk's blocks are trimmed to + * the changed core, anchored on context when either side would be empty, then + * re-padded with context lines until the old block occurs exactly once in the + * input; when the input has no room to disambiguate, the hunk is marked + * non-unique and prompts must state its start line. Blocks never start or end + * on a blank line when a visible context line is available — a blank line at a + * fence edge is invisible. Returns null when a hunk cannot be rendered at all. + */ +export function renderHunks(inputLines: string[], placements: Placement[]): RenderedHunk[] | null { + const merged = mergeNearbyPlacements(inputLines, placements, 2); + const hunks: RenderedHunk[] = []; + let previousEnd = -1; + + for (let index = 0; index < merged.length; index++) { + const placement = merged[index]; + const nextStart = index + 1 < merged.length ? merged[index + 1].start : inputLines.length; + const old = inputLines.slice(placement.start, placement.start + placement.oldLen); + const fresh = placement.newLines; + const prefix = commonPrefixLength(old, fresh); + const suffix = commonSuffixLength(old, fresh, Math.min(old.length, fresh.length) - prefix); + let shownStart = placement.start + prefix; + let oldBlock = old.slice(prefix, old.length - suffix); + let newBlock = fresh.slice(prefix, fresh.length - suffix); + if (oldBlock.length === 0 && newBlock.length === 0) continue; + + const takeAbove = (): boolean => { + const above = shownStart - 1; + if (above < 0 || above <= previousEnd) return false; + shownStart = above; + oldBlock = [inputLines[above], ...oldBlock]; + newBlock = [inputLines[above], ...newBlock]; + return true; + }; + const takeBelow = (): boolean => { + const below = shownStart + oldBlock.length; + if (below >= nextStart || below >= inputLines.length) return false; + oldBlock = [...oldBlock, inputLines[below]]; + newBlock = [...newBlock, inputLines[below]]; + return true; + }; + + // Pure insertion or pure deletion: anchor on a context line so neither + // block is empty. Prefer a non-blank anchor. + if (oldBlock.length === 0 || newBlock.length === 0) { + const aboveIndex = shownStart - 1; + const aboveVisible = aboveIndex >= 0 && aboveIndex > previousEnd && inputLines[aboveIndex].trim() !== ""; + if (aboveVisible) { + takeAbove(); + } else if (!takeBelow() && !takeAbove()) { + return null; + } + } + + let unique = true; + for (let extension = 0; findBlockOccurrences(inputLines, oldBlock).length > 1; extension++) { + if (extension >= MAX_UNIQUENESS_EXTENSION || !takeAbove()) { + unique = false; + break; + } + } + + // Pad blank fence edges with visible context (bounded; best effort). + for (let guard = 0; guard < 2 && (oldBlock[0].trim() === "" || newBlock[0]?.trim() === ""); guard++) { + if (!takeAbove()) break; + } + for ( + let guard = 0; + guard < 2 && (oldBlock[oldBlock.length - 1].trim() === "" || newBlock[newBlock.length - 1]?.trim() === ""); + guard++ + ) { + if (!takeBelow()) break; + } + + previousEnd = placement.start + placement.oldLen - 1; + hunks.push({ oldBlock, newBlock, startLine: shownStart + 1, unique }); + } + return hunks.length > 0 ? hunks : null; +} + +/** + * Re-solve the prompt from scratch: locate every hunk's old block in the input + * (unique occurrence, or the stated start line) and apply the replacements. + * Guarantees the prompt admits exactly one byte-exact answer. + */ +export function solveRenderedHunks(inputLines: string[], hunks: RenderedHunk[]): string[] | null { + const placements: Placement[] = []; + for (const hunk of hunks) { + const occurrences = findBlockOccurrences(inputLines, hunk.oldBlock); + let start: number; + if (hunk.unique) { + if (occurrences.length !== 1) return null; + start = occurrences[0]; + } else { + start = hunk.startLine - 1; + if (!occurrences.includes(start)) return null; + } + placements.push({ start, oldLen: hunk.oldBlock.length, newLines: hunk.newBlock }); + } + placements.sort((a, b) => a.start - b.start); + let previousEnd = -1; + for (const placement of placements) { + if (placement.start <= previousEnd) return null; + previousEnd = placement.start + placement.oldLen - 1; + } + return applyPlacements(inputLines, placements); +} + +export const LANGUAGE_BY_EXTENSION: Record = { + ".ts": "ts", + ".tsx": "tsx", + ".js": "js", + ".jsx": "jsx", + ".mjs": "js", + ".md": "markdown", + ".rs": "rust", + ".py": "python", + ".json": "json", + ".yml": "yaml", + ".yaml": "yaml", + ".toml": "toml", +}; + +/** Fence that safely wraps every block in `hunks` (upgrades when content embeds triple backticks). */ +export function pickFence(hunks: RenderedHunk[]): string { + const hasTripleBacktick = hunks.some(hunk => + [...hunk.oldBlock, ...hunk.newBlock].some(line => line.includes("```")), + ); + return hasTripleBacktick ? "````" : "```"; +} diff --git a/packages/typescript-edit-benchmark/src/mutations.ts b/packages/typescript-edit-benchmark/src/mutations.ts index c7eb1fb83..27d308e55 100644 --- a/packages/typescript-edit-benchmark/src/mutations.ts +++ b/packages/typescript-edit-benchmark/src/mutations.ts @@ -21,16 +21,18 @@ export interface MutationInfo { lineNumber: number; originalSnippet: string; mutatedSnippet: string; + /** Correct/misspelled rename pair; set by identifier-multi-edit for rename-style prompts. */ + identifier?: { correct: string; misspelled: string }; } export interface Mutation { name: string; category: string; - fixHint: string; + /** Whether this mutation intentionally produces multiple separated hunks. */ + multiHunk?: boolean; canApply(content: string): boolean; mutate(content: string, rng: () => number): [string, MutationInfo]; - describe(info: MutationInfo): string; } type Candidate = { @@ -234,12 +236,6 @@ function isLengthMemberExpression(node: t.Node): node is t.MemberExpression { abstract class BaseAstMutation implements Mutation { abstract name: string; abstract category: string; - abstract fixHint: string; - abstract description: string; - - describe(_info: MutationInfo): string { - return this.description; - } abstract collectCandidates(parsed: Parsed): Candidate[]; abstract applyCandidate(parsed: Parsed, candidate: Candidate, rng: () => number): MutationInfo; @@ -281,8 +277,6 @@ abstract class BaseAstMutation implements Mutation { class SwapComparisonMutation extends BaseAstMutation { name = "swap-comparison"; category = "operator"; - fixHint = "Swap the comparison operator to the correct variant."; - description = "A comparison operator is subtly wrong."; #swap: Record = { "<=": "<", @@ -310,8 +304,6 @@ class SwapComparisonMutation extends BaseAstMutation { class SwapEqualityMutation extends BaseAstMutation { name = "swap-equality"; category = "operator"; - fixHint = "Fix the equality comparison operator."; - description = "An equality operator is inverted."; #swap: Record = { "===": "!==", @@ -339,8 +331,6 @@ class SwapEqualityMutation extends BaseAstMutation { class SwapLogicalMutation extends BaseAstMutation { name = "swap-logical"; category = "operator"; - fixHint = "Use the intended boolean operator."; - description = "A boolean operator is incorrect."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -364,8 +354,6 @@ class SwapLogicalMutation extends BaseAstMutation { class RemoveNegationMutation extends BaseAstMutation { name = "remove-negation"; category = "operator"; - fixHint = "Add back the missing logical negation (`!`)."; - description = "A logical negation (`!`) was accidentally removed."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -393,8 +381,6 @@ class RemoveNegationMutation extends BaseAstMutation { class SwapIncDecMutation extends BaseAstMutation { name = "swap-increment-decrement"; category = "operator"; - fixHint = "Replace the increment/decrement operator with the intended one."; - description = "An increment/decrement operator points the wrong direction."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -417,8 +403,6 @@ class SwapIncDecMutation extends BaseAstMutation { class SwapArithmeticMutation extends BaseAstMutation { name = "swap-arithmetic"; category = "operator"; - fixHint = "Correct the arithmetic operator."; - description = "An arithmetic operator was swapped."; #swap: Record = { "+": "-", "-": "+", "*": "/", "/": "*" }; @@ -441,8 +425,6 @@ class SwapArithmeticMutation extends BaseAstMutation { class BooleanLiteralFlipMutation extends BaseAstMutation { name = "flip-boolean"; category = "literal"; - fixHint = "Flip the boolean literal to the intended value."; - description = "A boolean literal is inverted."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -465,9 +447,6 @@ class BooleanLiteralFlipMutation extends BaseAstMutation { class OptionalChainRemovalMutation extends BaseAstMutation { name = "remove-optional-chain"; category = "access"; - fixHint = - "Restore the optional chaining operator (`?.`) at the ONE location where it was removed. Do not add optional chaining elsewhere."; - description = "Optional chaining was removed from a property access."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -498,8 +477,6 @@ class OptionalChainRemovalMutation extends BaseAstMutation { class CallArgumentSwapMutation extends BaseAstMutation { name = "swap-call-args"; category = "call"; - fixHint = "Swap the two arguments to their original order."; - description = "Two arguments in a call are swapped."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -563,8 +540,6 @@ class CallArgumentSwapMutation extends BaseAstMutation { class NullishCoalescingSwapMutation extends BaseAstMutation { name = "swap-nullish"; category = "operator"; - fixHint = "Use the intended nullish/logical operator."; - description = "A nullish coalescing operator was swapped."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -588,8 +563,6 @@ class NullishCoalescingSwapMutation extends BaseAstMutation { class RegexQuantifierSwapMutation extends BaseAstMutation { name = "swap-regex-quantifier"; category = "regex"; - fixHint = "Fix the ONE regex quantifier that was swapped (between `+` and `*`). Do not modify other quantifiers."; - description = "A regex quantifier was swapped, changing whitespace matching."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -650,8 +623,6 @@ class RegexQuantifierSwapMutation extends BaseAstMutation { class UnicodeHyphenMutation extends BaseAstMutation { name = "unicode-hyphen"; category = "unicode"; - fixHint = "Replace the unicode dash with a plain ASCII hyphen."; - description = "A string literal contains a lookalike unicode dash."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -690,8 +661,7 @@ class UnicodeHyphenMutation extends BaseAstMutation { class IdentifierMultiEditMutation extends BaseAstMutation { name = "identifier-multi-edit"; category = "identifier"; - fixHint = "Restore the identifier to its original spelling in all affected locations."; - description = "An identifier is misspelled in multiple separate locations."; + multiHunk = true; #keywords = new Set([ "await", @@ -850,6 +820,7 @@ class IdentifierMultiEditMutation extends BaseAstMutation { lineNumber: selectedPaths[0]?.node.loc?.start.line ?? 0, originalSnippet: chosen.name, mutatedSnippet: mutated, + identifier: { correct: chosen.name, misspelled: mutated }, }, ]; } @@ -928,6 +899,7 @@ class IdentifierMultiEditMutation extends BaseAstMutation { lineNumber: selectedPaths[0]?.node.loc?.start.line ?? 0, originalSnippet: chosen.name, mutatedSnippet: mutated, + identifier: { correct: chosen.name, misspelled: mutated }, }; } } @@ -935,8 +907,6 @@ class IdentifierMultiEditMutation extends BaseAstMutation { class DuplicateLineLiteralFlipMutation extends BaseAstMutation { name = "duplicate-line-flip"; category = "duplicate"; - fixHint = "Fix the literal or operator on the duplicated line."; - description = "A duplicated line contains a subtle literal/operator change."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -1025,8 +995,6 @@ class DuplicateLineLiteralFlipMutation extends BaseAstMutation { class SwapAdjacentLinesMutation extends BaseAstMutation { name = "swap-adjacent-lines"; category = "structural"; - fixHint = "Swap the two adjacent lines back to their original order."; - description = "Two adjacent statements are in the wrong order."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -1134,8 +1102,6 @@ class SwapAdjacentLinesMutation extends BaseAstMutation { class SwapIfElseBranchesMutation extends BaseAstMutation { name = "swap-if-else"; category = "structural"; - fixHint = "Swap the if and else branch bodies back to their original positions."; - description = "The if and else branches are swapped."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -1163,29 +1129,31 @@ class SwapIfElseBranchesMutation extends BaseAstMutation { } } -class RemoveEarlyReturnMutation extends BaseAstMutation { - name = "remove-early-return"; +/** + * Remove one label from a fall-through `case A: case B:` pair. The fix inserts + * the missing label back — a deterministic pure-insertion task whose content + * is dictated by the visible sibling label, not by hidden deleted code. + */ +class RemoveCaseLabelMutation extends BaseAstMutation { + name = "remove-case-label"; category = "structural"; - fixHint = - "Restore the missing guard clause (if statement with early return). Add back the exact 3-line pattern: if condition, return statement, closing brace."; - description = "A guard clause (early return) was removed."; - collectCandidates(parsed: Parsed): Candidate[] { - const out: Candidate[] = []; + collectCandidates(parsed: Parsed): Candidate[] { + const out: Candidate[] = []; traverse(parsed.ast, { - IfStatement: path => { + SwitchCase: path => { const node = path.node; - if (node.alternate) return; - if (!t.isBlockStatement(node.consequent)) return; - if (node.consequent.body.length !== 1) return; - if (!t.isReturnStatement(node.consequent.body[0])) return; + if (!node.test || node.consequent.length > 0) return; + if (!t.isSwitchStatement(path.parent)) return; + const index = path.parent.cases.indexOf(node); + if (index < 0 || index >= path.parent.cases.length - 1) return; out.push({ path }); }, }); return out; } - applyCandidate(parsed: Parsed, candidate: Candidate): MutationInfo { + applyCandidate(parsed: Parsed, candidate: Candidate): MutationInfo { const node = candidate.path.node; const before = snippetFromSource(parsed.code, node, snippetFromNode(node)); candidate.path.remove(); @@ -1193,131 +1161,331 @@ class RemoveEarlyReturnMutation extends BaseAstMutation { } } -class SwapNamedImportsMutation extends BaseAstMutation { - name = "swap-named-imports"; - category = "import"; - fixHint = - "Swap ONLY the two imported names that are in the wrong order. Do not reorder other imports or modify other import statements."; - description = "Two named imports are swapped in a destructuring import."; +/** + * Wrap a block's body in a redundant `if (true) { ... }`. The fix removes the + * wrapper and dedents — an indentation-shifting multi-line edit whose entire + * content stays visible in the buggy file (no invisible-code reconstruction). + */ +class WrapRedundantIfMutation extends BaseAstMutation { + name = "wrap-redundant-if"; + category = "structural"; - collectCandidates(parsed: Parsed): Candidate[] { - const out: Candidate[] = []; + #lineSpan(node: t.Node): number { + if (!node.loc) return 0; + return node.loc.end.line - node.loc.start.line + 1; + } + + collectCandidates(parsed: Parsed): Candidate[] { + const out: Candidate[] = []; traverse(parsed.ast, { - ImportDeclaration: path => { - const named = path.node.specifiers - .map((spec, idx) => ({ spec, idx })) - .filter((entry): entry is { spec: t.ImportSpecifier; idx: number } => t.isImportSpecifier(entry.spec)) - .filter(({ spec }) => t.isIdentifier(spec.imported) && t.isIdentifier(spec.local)) - .filter( - ({ spec }) => - t.isIdentifier(spec.imported) && - t.isIdentifier(spec.local) && - spec.imported.name === spec.local.name, - ); - if (named.length < 2) return; - for (let i = 0; i < named.length; i++) { - for (let j = i + 1; j < named.length; j++) { - out.push({ path, meta: { i: named[i]!.idx, j: named[j]!.idx } }); - } - } + BlockStatement: path => { + const span = this.#lineSpan(path.node); + if (span < 4 || span > 160) return; + if (path.node.body.length === 0) return; + if (path.node.directives.length > 0) return; + // Don't wrap a body that is itself just an `if (true)` (repeated mutation noise). + if (path.node.body.length === 1 && t.isIfStatement(path.node.body[0])) return; + out.push({ path }); }, }); return out; } + applyCandidate(_parsed: Parsed, candidate: Candidate): MutationInfo { + const node = candidate.path.node; + const line = nodeLine(node); + node.body = [t.ifStatement(t.booleanLiteral(true), t.blockStatement(node.body))]; + return { lineNumber: line, originalSnippet: "", mutatedSnippet: "if (true) {" }; + } +} + +/** + * Swap two adjacent multi-line sibling statements (functions, if-chains, + * loops). The fix swaps them back — a large contiguous replace whose content + * is fully visible. + */ +class SwapSiblingBlocksMutation implements Mutation { + name = "swap-sibling-blocks"; + category = "structural"; + + #collect(parsed: Parsed): Array<{ left: t.Statement; right: t.Statement }> { + const out: Array<{ left: t.Statement; right: t.Statement }> = []; + const lineSpan = (node: t.Node): number => (node.loc ? node.loc.end.line - node.loc.start.line + 1 : 0); + + const considerList = (body: Array): void => { + for (let i = 0; i < body.length - 1; i++) { + const left = body[i]; + const right = body[i + 1]; + if (!left || !right || !t.isStatement(left) || !t.isStatement(right)) continue; + if (!left.loc || !right.loc) continue; + if (t.isImportDeclaration(left) || t.isImportDeclaration(right)) continue; + const leftSpan = lineSpan(left); + const rightSpan = lineSpan(right); + // At least one true block; bounded total so prompts stay readable. + if (Math.max(leftSpan, rightSpan) < 3) continue; + if (leftSpan > 80 || rightSpan > 80 || leftSpan + rightSpan > 120) continue; + const gap = right.loc.start.line - left.loc.end.line; + if (gap > 2) continue; + const leftText = snippetFromSource(parsed.code, left, "").trim(); + const rightText = snippetFromSource(parsed.code, right, "").trim(); + if (!leftText || !rightText || leftText === rightText) continue; + out.push({ left, right }); + } + }; + + traverse(parsed.ast, { + Program: path => considerList(path.node.body), + BlockStatement: path => considerList(path.node.body), + }); + return out; + } + + canApply(content: string): boolean { + const parsed = parseCode(content); + return parsed !== null && this.#collect(parsed).length > 0; + } + mutate(content: string, rng: () => number): [string, MutationInfo] { const parsed = parseCode(content); if (!parsed) return [content, noopInfo()]; - const candidates = this.collectCandidates(parsed); + const candidates = this.#collect(parsed); if (candidates.length === 0) return [content, noopInfo()]; - const chosen = randomChoice(candidates, rng); - const node = chosen.path.node; - const indices = chosen.meta; - if (!indices) return [content, noopInfo()]; - const { i, j } = indices; - if (i < 0 || j < 0 || i >= node.specifiers.length || j >= node.specifiers.length) return [content, noopInfo()]; - - const left = node.specifiers[i]; - const right = node.specifiers[j]; - if (!left || !right) return [content, noopInfo()]; + const { left, right } = randomChoice(candidates, rng); const leftRange = nodeRange(left); const rightRange = nodeRange(right); - const importRange = nodeRange(node); - if (!leftRange || !rightRange || !importRange) return [content, noopInfo()]; - if (leftRange.end > rightRange.start) return [content, noopInfo()]; + if (!leftRange || !rightRange || leftRange.end > rightRange.start) return [content, noopInfo()]; - const leftText = content.slice(leftRange.start, leftRange.end); - const rightText = content.slice(rightRange.start, rightRange.end); + const between = content.slice(leftRange.end, rightRange.start); + const swapped = `${content.slice(rightRange.start, rightRange.end)}${between}${content.slice(leftRange.start, leftRange.end)}`; const mutated = applySourceEdits(content, [ - { start: leftRange.start, end: leftRange.end, replacement: rightText }, - { start: rightRange.start, end: rightRange.end, replacement: leftText }, + { start: leftRange.start, end: rightRange.end, replacement: swapped }, ]); if (!mutated || mutated === content) return [content, noopInfo()]; return [ mutated, { - lineNumber: nodeLine(node), - originalSnippet: content.slice(importRange.start, importRange.end).trim(), - mutatedSnippet: mutated.slice(importRange.start, importRange.end).trim(), + lineNumber: left.loc?.start.line ?? 0, + originalSnippet: `lines ${left.loc?.start.line ?? 0}-${right.loc?.end.line ?? 0}`, + mutatedSnippet: "[swapped]", }, ]; } - - applyCandidate(parsed: Parsed, candidate: Candidate): MutationInfo { - const node = candidate.path.node; - const indices = candidate.meta; - if (!indices) return noopInfo(); - const before = snippetFromSource(parsed.code, node, snippetFromNode(node)); - const { i, j } = indices; - if (i < 0 || j < 0 || i >= node.specifiers.length || j >= node.specifiers.length) return noopInfo(); - [node.specifiers[i], node.specifiers[j]] = [node.specifiers[j]!, node.specifiers[i]!]; - return { - lineNumber: nodeLine(node), - originalSnippet: before.trim(), - mutatedSnippet: snippetFromNode(node).trim(), - }; - } } -class DeleteStatementMutation extends BaseAstMutation { - name = "delete-statement"; +/** + * Duplicate a multi-line statement right after itself (a copy-paste accident). + * The fix deletes the second copy — a large deletion where the surviving copy + * stays visible, so the task is deterministic without revealing hidden code. + */ +class DuplicateBlockMutation implements Mutation { + name = "duplicate-block"; category = "structural"; - fixHint = "Restore the deleted statement."; - description = "A critical statement was deleted from the code."; - collectCandidates(parsed: Parsed): Candidate[] { - const out: Candidate[] = []; + #collect(parsed: Parsed): t.Statement[] { + const out: t.Statement[] = []; + const consider = (statement: t.Statement | t.ModuleDeclaration): void => { + if (!t.isStatement(statement) || !statement.loc) return; + const span = statement.loc.end.line - statement.loc.start.line + 1; + if (span < 3 || span > 120) return; + // Duplicating lexical declarations or exports produces parse/redeclaration + // errors; stick to statements that stay syntactically valid twice. + if ( + !t.isIfStatement(statement) && + !t.isExpressionStatement(statement) && + !t.isForStatement(statement) && + !t.isForOfStatement(statement) && + !t.isWhileStatement(statement) && + !t.isTryStatement(statement) && + !t.isSwitchStatement(statement) + ) { + return; + } + out.push(statement); + }; traverse(parsed.ast, { - Statement: path => { - if (!path.node.loc) return; - if (t.isVariableDeclaration(path.node)) { - out.push({ path: path as NodePath }); - return; - } - if (!t.isExpressionStatement(path.node)) return; - if (t.isAssignmentExpression(path.node.expression) || t.isUpdateExpression(path.node.expression)) { - out.push({ path: path as NodePath }); - } + Program: path => { + for (const statement of path.node.body) consider(statement); + }, + BlockStatement: path => { + for (const statement of path.node.body) consider(statement); }, }); return out; } - applyCandidate(parsed: Parsed, candidate: Candidate): MutationInfo { - const node = candidate.path.node; - const before = snippetFromSource(parsed.code, node, snippetFromNode(node)); - candidate.path.remove(); - return { lineNumber: nodeLine(node), originalSnippet: before.trim(), mutatedSnippet: "[removed]" }; + canApply(content: string): boolean { + const parsed = parseCode(content); + return parsed !== null && this.#collect(parsed).length > 0; + } + + mutate(content: string, rng: () => number): [string, MutationInfo] { + const parsed = parseCode(content); + if (!parsed) return [content, noopInfo()]; + const candidates = this.#collect(parsed); + if (candidates.length === 0) return [content, noopInfo()]; + + const statement = randomChoice(candidates, rng); + const range = nodeRange(statement); + if (!range) return [content, noopInfo()]; + + const text = content.slice(range.start, range.end); + const mutated = applySourceEdits(content, [{ start: range.end, end: range.end, replacement: `\n\n${text}` }]); + if (!mutated || mutated === content) return [content, noopInfo()]; + // Reject mutations that no longer parse (e.g. duplicated declarations). + if (!parseCode(mutated)) return [content, noopInfo()]; + + return [ + mutated, + { + lineNumber: statement.loc?.start.line ?? 0, + originalSnippet: text.split("\n")[0]?.trim() ?? "", + mutatedSnippet: text.split("\n")[0]?.trim() ?? "", + }, + ]; + } +} + +/** + * Move a multi-line statement to a distant position in the same statement + * list. The fix moves it back — one delete hunk plus one insert hunk, the two + * dominant hunk shapes in real edits — with the moved content fully visible. + */ +class MoveDistantBlockMutation implements Mutation { + name = "move-distant-block"; + category = "structural"; + multiHunk = true; + + #collect(parsed: Parsed): Array<{ moved: t.Statement; next: t.Statement; target: t.Statement }> { + const out: Array<{ moved: t.Statement; next: t.Statement; target: t.Statement }> = []; + + const considerList = (body: Array): void => { + if (body.length < 5) return; + for (let from = 0; from < body.length - 1; from++) { + const moved = body[from]; + const next = body[from + 1]; + if (!moved || !next || !t.isStatement(moved) || !t.isStatement(next)) continue; + if (!moved.loc || t.isImportDeclaration(moved)) continue; + const span = moved.loc.end.line - moved.loc.start.line + 1; + if (span < 3 || span > 60) continue; + for (let to = from + 3; to < body.length; to++) { + const target = body[to]; + if (!target || !t.isStatement(target) || !target.loc) continue; + out.push({ moved, next, target }); + } + } + }; + + traverse(parsed.ast, { + Program: path => considerList(path.node.body), + BlockStatement: path => considerList(path.node.body), + }); + return out; + } + + canApply(content: string): boolean { + const parsed = parseCode(content); + return parsed !== null && this.#collect(parsed).length > 0; + } + + mutate(content: string, rng: () => number): [string, MutationInfo] { + const parsed = parseCode(content); + if (!parsed) return [content, noopInfo()]; + const candidates = this.#collect(parsed); + if (candidates.length === 0) return [content, noopInfo()]; + + const { moved, next, target } = randomChoice(candidates, rng); + const movedRange = nodeRange(moved); + const nextRange = nodeRange(next); + const targetRange = nodeRange(target); + if (!movedRange || !nextRange || !targetRange) return [content, noopInfo()]; + if (movedRange.end > nextRange.start || nextRange.start > targetRange.end) return [content, noopInfo()]; + + const movedText = content.slice(movedRange.start, movedRange.end); + const mutated = applySourceEdits(content, [ + // Cut the statement together with its trailing separator... + { start: movedRange.start, end: nextRange.start, replacement: "" }, + // ...and splice it back in after the distant target statement. + { start: targetRange.end, end: targetRange.end, replacement: `\n\n${movedText}` }, + ]); + if (!mutated || mutated === content) return [content, noopInfo()]; + if (!parseCode(mutated)) return [content, noopInfo()]; + + return [ + mutated, + { + lineNumber: moved.loc?.start.line ?? 0, + originalSnippet: movedText.split("\n")[0]?.trim() ?? "", + mutatedSnippet: movedText.split("\n")[0]?.trim() ?? "", + }, + ]; + } +} + +/** + * Apply several independent token-level mutations in one file — the multi-hunk + * edit shape that dominates real sessions. Each constituent bug is a + * single-line change fully specified by the task's before/after blocks. + */ +class CompositeMultiEditMutation implements Mutation { + name = "composite-multi-edit"; + category = "multi"; + multiHunk = true; + + #parts: Mutation[]; + + constructor(parts: Mutation[]) { + this.#parts = parts; + } + + canApply(content: string): boolean { + let applicable = 0; + for (const part of this.#parts) { + try { + if (part.canApply(content)) applicable++; + } catch { + // Unparseable for this part; skip. + } + if (applicable >= 2) return true; + } + return false; + } + + mutate(content: string, rng: () => number): [string, MutationInfo] { + const applicable = this.#parts.filter(part => { + try { + return part.canApply(content); + } catch { + return false; + } + }); + if (applicable.length < 2) return [content, noopInfo()]; + + const target = Math.min(3 + Math.floor(rng() * 3), applicable.length); + const parts = randomSample(applicable, target, rng); + let current = content; + let firstInfo: MutationInfo | null = null; + let applied = 0; + for (const part of parts) { + try { + const [next, info] = part.mutate(current, rng); + if (next === current || info.lineNumber === 0) continue; + current = next; + firstInfo ??= info; + applied++; + } catch { + // A part failing on already-mutated content just shrinks the composite. + } + } + if (applied < 2 || !firstInfo) return [content, noopInfo()]; + return [current, firstInfo]; } } class OffByOneMutation extends BaseAstMutation { name = "off-by-one"; category = "literal"; - fixHint = "Fix the off-by-one error in the numeric literal or comparison."; - description = "A numeric boundary has an off-by-one error."; collectCandidates(parsed: Parsed): Candidate[] { const out: Candidate[] = []; @@ -1386,7 +1554,8 @@ class OffByOneMutation extends BaseAstMutation { } } -export const ALL_MUTATIONS: Mutation[] = [ +/** Single-line token mutations; also the constituent parts of the composite. */ +const TOKEN_MUTATIONS: Mutation[] = [ new SwapComparisonMutation(), new SwapEqualityMutation(), new SwapLogicalMutation(), @@ -1399,14 +1568,21 @@ export const ALL_MUTATIONS: Mutation[] = [ new NullishCoalescingSwapMutation(), new RegexQuantifierSwapMutation(), new UnicodeHyphenMutation(), + new OffByOneMutation(), +]; + +export const ALL_MUTATIONS: Mutation[] = [ + ...TOKEN_MUTATIONS, new IdentifierMultiEditMutation(), new DuplicateLineLiteralFlipMutation(), new SwapAdjacentLinesMutation(), new SwapIfElseBranchesMutation(), - new RemoveEarlyReturnMutation(), - new SwapNamedImportsMutation(), - new DeleteStatementMutation(), - new OffByOneMutation(), + new WrapRedundantIfMutation(), + new SwapSiblingBlocksMutation(), + new DuplicateBlockMutation(), + new MoveDistantBlockMutation(), + new RemoveCaseLabelMutation(), + new CompositeMultiEditMutation(TOKEN_MUTATIONS), ]; export const CATEGORY_MAP: Record = { @@ -1419,5 +1595,5 @@ export const CATEGORY_MAP: Record = { identifier: ALL_MUTATIONS.filter(m => m.category === "identifier").map(m => m.name), duplicate: ALL_MUTATIONS.filter(m => m.category === "duplicate").map(m => m.name), structural: ALL_MUTATIONS.filter(m => m.category === "structural").map(m => m.name), - import: ALL_MUTATIONS.filter(m => m.category === "import").map(m => m.name), + multi: ALL_MUTATIONS.filter(m => m.category === "multi").map(m => m.name), }; diff --git a/packages/typescript-edit-benchmark/src/prompts/identifier-task.md b/packages/typescript-edit-benchmark/src/prompts/identifier-task.md new file mode 100644 index 000000000..34c0e1515 --- /dev/null +++ b/packages/typescript-edit-benchmark/src/prompts/identifier-task.md @@ -0,0 +1,9 @@ +# Fix a misspelled identifier in `{{filename}}` + +A recent edit misspelled the identifier `{{correct}}` as `{{misspelled}}` in {{#when count "==" 1}}one place{{else}}{{count}} places{{/when}}. +{{#if affectedLines}} + +Affected line{{#when count ">" 1}}s{{/when}}: {{join affectedLines ", "}}. +{{/if}} + +Replace every occurrence of `{{misspelled}}` with `{{correct}}`. Do not change anything else. diff --git a/packages/typescript-edit-benchmark/src/prompts/mutation-task.md b/packages/typescript-edit-benchmark/src/prompts/mutation-task.md new file mode 100644 index 000000000..e009de7e7 --- /dev/null +++ b/packages/typescript-edit-benchmark/src/prompts/mutation-task.md @@ -0,0 +1,85 @@ +# Fix a bug in `{{filename}}` + +{{#when name "==" "swap-comparison"}} +A comparison operator is subtly wrong. +{{/when}} +{{#when name "==" "swap-equality"}} +An equality operator is inverted. +{{/when}} +{{#when name "==" "swap-logical"}} +A boolean operator is incorrect. +{{/when}} +{{#when name "==" "remove-negation"}} +A logical negation (`!`) was accidentally removed. +{{/when}} +{{#when name "==" "swap-increment-decrement"}} +An increment/decrement operator points the wrong direction. +{{/when}} +{{#when name "==" "swap-arithmetic"}} +An arithmetic operator was swapped. +{{/when}} +{{#when name "==" "flip-boolean"}} +A boolean literal is inverted. +{{/when}} +{{#when name "==" "remove-optional-chain"}} +Optional chaining was removed from a property access. +{{/when}} +{{#when name "==" "swap-call-args"}} +Two arguments in a call are swapped. +{{/when}} +{{#when name "==" "swap-nullish"}} +A nullish coalescing operator was swapped. +{{/when}} +{{#when name "==" "swap-regex-quantifier"}} +A regex quantifier was swapped, changing whitespace matching. +{{/when}} +{{#when name "==" "unicode-hyphen"}} +A string literal contains a lookalike unicode dash. +{{/when}} +{{#when name "==" "off-by-one"}} +A numeric boundary has an off-by-one error. +{{/when}} +{{#when name "==" "duplicate-line-flip"}} +One copy of a line that repeats elsewhere in this file was altered. +{{/when}} +{{#when name "==" "composite-multi-edit"}} +This file contains several small, unrelated single-line bugs. +{{/when}} +{{#if functionName}} + +The bug is in the `{{functionName}}` function. +{{/if}} +{{#when region "==" "top"}} + +The bug is near the top of the file. +{{/when}} +{{#when region "==" "middle"}} + +The bug is around the middle of the file. +{{/when}} +{{#when region "==" "end"}} + +The bug is near the end of the file. +{{/when}} +{{#each hunks}} +{{#when ../hunkCount ">" 1}} + +## Change {{add @index 1}} +{{/when}} + +{{#if isDelete}}Delete this block{{#if startLine}} (starting on line {{startLine}}){{/if}}:{{else}}Replace this{{#if startLine}} (starting on line {{startLine}}){{/if}}:{{/if}} + +{{../fence}}{{../language}} +{{oldCode}} +{{../fence}} +{{#unless isDelete}} + +with: + +{{../fence}}{{../language}} +{{newCode}} +{{../fence}} +{{/unless}} +{{/each}} + +{{#if nightmare}}This file contains near-identical code in multiple places — edit exactly the block shown and nothing else.{{else}}Make exactly this change; do not modify anything else.{{/if}} diff --git a/packages/typescript-edit-benchmark/src/prompts/structural-task.md b/packages/typescript-edit-benchmark/src/prompts/structural-task.md new file mode 100644 index 000000000..3792f45eb --- /dev/null +++ b/packages/typescript-edit-benchmark/src/prompts/structural-task.md @@ -0,0 +1,37 @@ +# Fix a bug in `{{filename}}` + +{{#when kind "==" "case-label"}} +In this file's `switch`, the value in `{{label}}` must be handled exactly like the case after it: add a fall-through `{{label}}` label directly before `{{before}}`. +{{/when}} +{{#when kind "==" "duplicate-block"}} +The block starting with `{{head}}` appears twice in a row — the second copy is a copy-paste accident. Delete the second copy and keep the first. +{{/when}} +{{#when kind "==" "move-block"}} +The block starting with `{{head}}` was moved to the wrong place — it currently sits after `{{currentPrev}}`. Move it back so it comes directly before `{{destination}}`. +{{/when}} +{{#when kind "==" "wrap-if"}} +A leftover debugging wrapper is redundant: remove the `if (true) {` on line {{wrapperLine}} together with its closing brace, and dedent the wrapped body one level. +{{/when}} +{{#when kind "==" "swap-blocks"}} +Two adjacent blocks are in the wrong order: the block starting with `{{secondHead}}` belongs before the block starting with `{{firstHead}}`. Swap the two blocks. +{{/when}} +{{#when kind "==" "swap-lines"}} +Two adjacent statements are in the wrong order: `{{secondHead}}` belongs before `{{firstHead}}`. Swap the two statements. +{{/when}} +{{#when kind "==" "swap-if-else"}} +The branch bodies of `{{condition}}` are swapped: the current `else` body belongs under the `if`, and vice versa. Swap the two branch bodies. +{{/when}} + +After the fix, the affected {{#when hunkCount ">" 1}}regions must{{else}}region must{{/when}} read exactly: +{{#each hunks}} +{{#if startLine}} + +Around line {{startLine}}: +{{/if}} + +{{../fence}}{{../language}} +{{newCode}} +{{../fence}} +{{/each}} + +Make exactly this change; do not modify anything else. diff --git a/packages/typescript-edit-benchmark/test/hunks.test.ts b/packages/typescript-edit-benchmark/test/hunks.test.ts new file mode 100644 index 000000000..4518da59b --- /dev/null +++ b/packages/typescript-edit-benchmark/test/hunks.test.ts @@ -0,0 +1,96 @@ +import { describe, expect, it } from "bun:test"; +import { + placementsFromDiff, + type RenderedHunk, + renderHunks, + solveRenderedHunks, +} from "@oh-my-pi/typescript-edit-benchmark/hunks"; + +/** Round-trip an input/expected pair through the prompt machinery. */ +function roundTrip(input: string, expected: string): { hunks: RenderedHunk[]; solved: string } { + const inputLines = input.replace(/\n$/, "").split("\n"); + const placements = placementsFromDiff(input, expected); + const hunks = renderHunks(inputLines, placements); + if (!hunks) throw new Error("renderHunks returned null"); + const solved = solveRenderedHunks(inputLines, hunks); + if (!solved) throw new Error("solveRenderedHunks returned null"); + return { hunks, solved: `${solved.join("\n")}\n` }; +} + +describe("hunk round-trip", () => { + it("reproduces the expected file for a block replacement", () => { + const input = "const a = 1;\nfunction f() {\n\treturn a;\n}\nexport { f };\n"; + const expected = "const a = 1;\nfunction f() {\n\treturn a + 1;\n}\nexport { f };\n"; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + expect(hunks).toHaveLength(1); + expect(hunks[0].unique).toBe(true); + }); + + it("reproduces the expected file for a pure insertion, anchored on a non-blank line", () => { + const input = "function f() {\n\n\treturn 1;\n}\n"; + const expected = "function f() {\n\n\tconst x = 0;\n\treturn 1;\n}\n"; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + // The line directly above the insertion is blank; the anchor must be visible. + expect(hunks[0].oldBlock.some(line => line.trim().length > 0)).toBe(true); + }); + + it("reproduces the expected file for a pure deletion, anchored on context", () => { + const input = "a();\nb();\nc();\n"; + const expected = "a();\nc();\n"; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + // Deletions keep a context anchor so the replacement is never an empty fence. + expect(hunks[0].newBlock.length).toBeGreaterThan(0); + expect(hunks[0].oldBlock.length).toBeGreaterThan(hunks[0].newBlock.length); + }); +}); + +describe("uniqueness handling", () => { + it("extends a repeated block with context until it is unique", () => { + const input = "start();\nif (x) {\n\tstop();\n}\nif (y) {\n\tstop();\n}\n"; + const expected = "start();\nif (x) {\n\tstop();\n}\nif (y) {\n\thalt();\n}\n"; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + expect(hunks[0].unique).toBe(true); + // `\tstop();` alone is ambiguous; the block must carry the `if (y) {` context. + expect(hunks[0].oldBlock.length).toBeGreaterThan(1); + }); + + it("marks a hunk non-unique with its start line when context cannot disambiguate", () => { + const line = "value += 1;"; + const input = `${Array.from({ length: 4 }, () => line).join("\n")}\n`; + const expected = `${[line, line, "value += 2;", line].join("\n")}\n`; + const inputLines = input.replace(/\n$/, "").split("\n"); + const hunks = renderHunks(inputLines, placementsFromDiff(input, expected)); + if (!hunks) throw new Error("renderHunks returned null"); + expect(hunks[0].unique).toBe(false); + // startLine must locate a real occurrence of the (extended) old block, so + // a solver that trusts it lands on the right copy. + expect(inputLines.slice(hunks[0].startLine - 1, hunks[0].startLine - 1 + hunks[0].oldBlock.length)).toEqual( + hunks[0].oldBlock, + ); + const solved = solveRenderedHunks(inputLines, hunks); + expect(solved ? `${solved.join("\n")}\n` : null).toBe(expected); + }); +}); + +describe("placement merging", () => { + it("renders two adjacent swapped statements as a single hunk", () => { + const input = "setup();\nsecond();\nfirst();\nteardown();\n"; + const expected = "setup();\nfirst();\nsecond();\nteardown();\n"; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + expect(hunks).toHaveLength(1); + }); + + it("keeps distant changes as separate hunks", () => { + const filler = Array.from({ length: 10 }, (_, i) => `line${i}();`).join("\n"); + const input = `alpha();\n${filler}\nomega();\n`; + const expected = `alpha2();\n${filler}\nomega2();\n`; + const { hunks, solved } = roundTrip(input, expected); + expect(solved).toBe(expected); + expect(hunks).toHaveLength(2); + }); +});