feat(typescript-edit-benchmark): introduced empirical edit mutation planning

- Add new structural, multi-edit, and block-level mutation classes with updated category mappings.
- Introduce hunk extraction, placement, rendering, and solver utilities along with unit tests.
- Implement size-based mutation planning, prompt validation logic, and new prompt markdown templates.
- Update benchmark generation scripts and package configurations to support empirical edit shape statistics.
This commit is contained in:
can1357
2026-07-30 07:43:15 +02:00
parent fcddf46e12
commit 9856f904d7
18 changed files with 1543 additions and 229 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Breaking Changes
- Removed the `DEL`, `DEL.BLK`, `COPY`, and `COPY.BLK` hashline edit operations. Use `CUT` / `CUT.BLK` for deletion; removed content remains available to `PASTE`.
### Added
- Added server-name autocomplete for `/mcp` commands (`enable`, `disable`, `test`, `remove`, `reconnect`, `reauth`, `unauth`) using configured and runtime-discovered MCP servers.
@@ -210,7 +210,6 @@ function renderSection(
};
}
export async function executeHashlineSingle(
options: ExecuteHashlineSingleOptions,
): Promise<AgentToolResult<EditToolDetails, typeof hashlineEditParamsSchema>> {
+5 -1
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Breaking Changes
- Removed `DEL`, `DEL.BLK`, `COPY`, and `COPY.BLK` from the patch language. Use `CUT` / `CUT.BLK` for deletion; a cut does not require a following `PASTE` and leaves the removed content available to later pastes.
### Added
- Added clipboard ops: `CUT N.=M` captures lines into a register (and deletes them), `CUT.BLK N` captures tree-sitter blocks, and `PASTE.PRE|POST N` / `PASTE.HEAD|TAIL` / `PASTE.BLK.POST N` insert the captured lines without retyping. The register flows top-to-bottom across sections, so content moves between files in one patch; `PASTE` does not consume it and the last capture wins.
@@ -10,7 +14,7 @@
### Changed
- Regrouped `grammar.lark` around shared shapes — one `target` rule (`N.=M` | `.BLK N`) for `DEL`/`CUT` and one `pos` rule (`PRE`/`POST`/`BLK.POST`/`HEAD`/`TAIL`) for `INS`/`PASTE` — collapsing per-op hunk rules. The accepted language is byte-identical.
- Simplified `grammar.lark` around shared target and position shapes, collapsing the concrete and block `CUT` forms plus the `INS` / `PASTE` position variants into their common grammar rules.
### Fixed
-1
View File
@@ -314,7 +314,6 @@ export class PatchSection {
});
}
/** Anchor lines touched by this section, sorted ascending and deduplicated. */
collectAnchorLines(): readonly number[] {
const lines = new Set<number>();
-2
View File
@@ -263,7 +263,6 @@ export function ambiguousCloserSpareMessage(
export const UNRESOLVED_BLOCK_INTERNAL =
"internal error: unresolved block edit reached the applier (resolveBlockEdits was not run).";
/** `REM` received a body row or coexists with line edits. */
export const REM_TAKES_NO_BODY =
"`REM` deletes the whole file and takes no body rows or line ops. Issue it alone under the header.";
@@ -272,7 +271,6 @@ export const REM_TAKES_NO_BODY =
export const MOVE_TAKES_NO_BODY =
"`MV DEST` does not take body rows. Put line edits above the `MV` row; the destination path follows `MV` on the same line.";
/** `CUT N.=M` hunk received a body row. */
export const CUT_TAKES_NO_BODY = `\`CUT N${HL_RANGE_SEP}M\` captures + deletes lines and takes no body rows. To replace lines with new content, use \`SWAP N${HL_RANGE_SEP}M:\`.`;
-1
View File
@@ -178,7 +178,6 @@ describe("clipboard across sections and batches", () => {
expect(fs.get("b.ts")).toBe("b1\nmove1\nmove2\n");
});
it("applies a batch-local CUT without requiring PASTE", async () => {
const a = "l1\nl2\n";
const { fs, patcher, tags } = taggedPatcher([["a.ts", a]]);
+3 -2
View File
@@ -52,8 +52,9 @@ describe("hashline format v4", () => {
});
it("does not recognize removed DEL or COPY headers", () => {
expect(() => parsePatch("DEL 2")).toThrow(/payload line has no preceding hunk header/);
expect(() => parsePatch("COPY 2")).toThrow(/payload line has no preceding hunk header/);
for (const header of ["DEL 2", "DEL.BLK 2", "COPY 2", "COPY.BLK 2"]) {
expect(() => parsePatch(header)).toThrow(/payload line has no preceding hunk header/);
}
});
it("auto-pipes bare body rows as literal text", () => {
@@ -3,13 +3,12 @@ You are participating in a code-edit benchmark inside a repository with {{#if mu
This benchmark is scored on exactness. Get the edit right.
## Important constraints
- Make the minimum change necessary. Do not refactor, improve, or clean up other code.
- If you see multiple similar patterns, only change the ONE that is buggy (there is only one intended mutation).
- Preserve exact code structure. Do not rearrange statements or change formatting.
- Make exactly the change the task specifies — nothing more. Do not refactor, improve, or clean up other code.
- Tasks range from single-token fixes to multi-hunk block rewrites. When the task shows replacement code, reproduce it byte-for-byte: indentation, tabs vs spaces, and blank lines included.
- If the file contains multiple similar regions, change only the one(s) the task identifies.
- Your output is verified by exact text diff against an expected fixture. Equivalent code, reordered imports, reordered object keys, or formatting changes will fail.
- Prefer copying the original line(s) and changing only the specific token(s) required. Do not rewrite whole statements.
- Never modify comments or license headers unless the task explicitly asks.
- Re-read the changed region after editing to confirm you only touched the intended line(s).
- Re-read the changed region after editing to confirm it matches the task exactly.
{{#if multiFile}}- Only modify the file(s) referenced by the task or follow-up messages. Leave all other files unchanged.
{{/if}}
## Process
Binary file not shown.
@@ -26,7 +26,8 @@
"test": "bun test --parallel",
"fix": "biome check --write --unsafe .",
"fmt": "biome format --write .",
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4"
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source",
"edit-shapes": "bun run src/edit-shape-stats.ts"
},
"dependencies": {
"@babel/generator": "catalog:",
+317
View File
@@ -0,0 +1,317 @@
#!/usr/bin/env bun
/**
* Measure the empirical shape of real agent edits from session logs.
*
* Scans session JSONL files for successful `edit`/`apply_patch` tool calls and
* parses the diff in each tool result (both the numbered ` NNN|` format and
* unified `@@` diffs) into change runs — maximal runs of `-`/`+` rows with no
* context row between them — so unchanged context never counts as changed and
* every run counts as its own hunk.
*
* Shapes are reported at two levels:
* - per tool call — one sample per edit call;
* - per user request — all edit calls between one user message and the next
* grouped into one sample (one benchmark fixture ≙ one prompt, and the
* runner allows multiple edit calls per prompt). Retried/overlapping edits
* within a request are summed, so request sizes lean high, and a request
* spans every file a long agentic turn touched — an upper bound for
* single-file fixtures.
*
* Parse coverage is printed so a format skew cannot silently bias the numbers.
* Reference measurement (2026-07, 2,000 newest sessions against this repo,
* 99.9% coverage):
*
* per request: changed lines: 1 → 6% 2-5 → 11% 6-20 → 20% 21-60 → 25% 61+ → 39%
* hunks: 1 → 11% 2 → 8% 3+ → 81% (median 40 changed lines)
* per call: changed lines: 1 → 23% 2-5 → 30% 6-20 → 29% 21-60 → 14% 61+ → 5%
* hunks: 1 → 48% 2 → 21% 3+ → 31% (median 5 changed lines)
* op mix (both levels): replace 55% insert 32% delete 12%
*
* `generate.ts` fixtures are single-file, single-prompt tasks: the suite is
* calibrated to sit between the two levels — per-call shapes for token fixes,
* request-leaning shapes (large blocks, multi-hunk composites) for the rest.
*/
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { parseArgs } from "node:util";
import { computeDefaultSessionDir } from "@oh-my-pi/pi-coding-agent/session/session-paths";
import { FileSessionStorage } from "@oh-my-pi/pi-coding-agent/session/session-storage";
const EDIT_TOOL_NAMES: Record<string, true> = { edit: true, apply_patch: true };
const SIZE_BUCKETS: Array<[string, number]> = [
["1", 1],
["2-5", 5],
["6-20", 20],
["21-60", 60],
["61+", Number.POSITIVE_INFINITY],
];
/** One maximal run of removed/added rows with no context row in between. */
interface ChangeRun {
oldCount: number;
newCount: number;
}
interface Args {
sessionsDir: string;
maxSessions: number;
}
function parseArguments(): Args {
const { values } = parseArgs({
options: {
"sessions-dir": { type: "string" },
repo: { type: "string", default: path.resolve(import.meta.dir, "../../..") },
"max-sessions": { type: "string", default: "2000" },
},
});
const repo = path.resolve(values.repo ?? ".");
return {
sessionsDir: values["sessions-dir"] ?? computeDefaultSessionDir(repo, new FileSessionStorage()),
maxSessions: parseInt(values["max-sessions"] ?? "2000", 10),
};
}
const ROW_PIPE_RE = /^([ +-])\s*\d+\|/;
const ROW_SPACE_RE = /^([ +-])\s*\d+( |$)/;
/** Collects change runs; a context row or elision closes the current run. */
class RunCollector {
runs: ChangeRun[] = [];
#current: ChangeRun | null = null;
mark(kind: string): void {
if (kind === " ") {
this.flush();
return;
}
this.#current ??= { oldCount: 0, newCount: 0 };
if (kind === "-") this.#current.oldCount++;
else this.#current.newCount++;
}
flush(): void {
if (this.#current && (this.#current.oldCount > 0 || this.#current.newCount > 0)) {
this.runs.push(this.#current);
}
this.#current = null;
}
finish(): ChangeRun[] | null {
this.flush();
return this.runs.length > 0 ? this.runs : null;
}
}
/**
* Parse the numbered diff format emitted in edit tool-result details
* (` NNN|context`, `-NNN|removed`, `+NNN|added`; older space-delimited rows).
* Elisions appear as blank rows or `...` rows. Returns null when any row is
* unparseable.
*/
function runsFromNumberedDiff(diff: string): ChangeRun[] | null {
const lines = diff.split("\n");
const usesPipe = lines.slice(0, 10).some(line => ROW_PIPE_RE.test(line));
const collector = new RunCollector();
for (const line of lines) {
const row = ROW_PIPE_RE.exec(line) ?? (usesPipe ? null : ROW_SPACE_RE.exec(line));
if (row) {
collector.mark(row[1]);
} else if (line.trim() === "" || line.trim() === "..." || line.trim() === "…" || line.trim().endsWith("|...")) {
collector.flush();
} else {
return null;
}
}
return collector.finish();
}
/**
* Parse a unified diff (standard `@@ -a,b +c,d @@` hunks or anchor-style `@@`
* headers): `-` old, `+` new, leading space or blank = context.
*/
function runsFromUnifiedDiff(diff: string): ChangeRun[] | null {
const collector = new RunCollector();
for (const line of diff.split("\n")) {
if (line.startsWith("@@")) {
collector.flush();
} else if (line.startsWith("--- ") || line.startsWith("+++ ") || line.startsWith("Index:")) {
// File headers.
} else if (line.startsWith("-") || line.startsWith("+")) {
collector.mark(line[0]);
} else if (line.startsWith(" ") || line === "") {
collector.mark(" ");
} else {
return null;
}
}
return collector.finish();
}
/** Parse a tool-result diff in whichever format it uses. */
function runsFromDiff(diff: string): { runs: ChangeRun[] | null; format: "numbered" | "unified" } {
const head = diff.trimStart();
if (head.startsWith("@@") || head.startsWith("--- ")) {
return { runs: runsFromUnifiedDiff(diff), format: "unified" };
}
return { runs: runsFromNumberedDiff(diff), format: "numbered" };
}
/** Per-session parse results: change runs per edit call and per user request, plus coverage counters. */
interface SessionScan {
edits: ChangeRun[][];
/** All edit-call runs between one user message and the next, concatenated. */
requests: ChangeRun[][];
formats: Map<string, number>;
skipped: number;
}
/** Extract change runs for every successful edit in one session JSONL file. */
async function collectSessionEdits(sessionFile: string): Promise<SessionScan> {
const scan: SessionScan = { edits: [], requests: [], formats: new Map(), skipped: 0 };
const stream = Bun.file(sessionFile).stream();
const decoder = new TextDecoder();
let buffer = "";
let currentRequest: ChangeRun[] = [];
const flushRequest = (): void => {
if (currentRequest.length > 0) scan.requests.push(currentRequest);
currentRequest = [];
};
const handleRecord = (record: unknown): void => {
if (!record || typeof record !== "object") return;
const rec = record as Record<string, unknown>;
if (rec.type !== "message" || !rec.message || typeof rec.message !== "object") return;
const message = rec.message as Record<string, unknown>;
// A user message starts a new request; everything until the next one
// belongs to the same prompt (one benchmark fixture ≙ one request).
if (message.role === "user") {
flushRequest();
return;
}
if (message.role !== "toolResult" || !EDIT_TOOL_NAMES[String(message.toolName)] || message.isError) return;
const details =
message.details && typeof message.details === "object"
? (message.details as Record<string, unknown>)
: undefined;
const diff = details && typeof details.diff === "string" ? details.diff : null;
const op = details && typeof details.op === "string" ? details.op : null;
if (!diff || (op && op !== "update")) return;
const { runs, format } = runsFromDiff(diff);
if (runs) {
scan.edits.push(runs);
currentRequest.push(...runs);
scan.formats.set(format, (scan.formats.get(format) ?? 0) + 1);
} else {
scan.skipped++;
}
};
for await (const chunk of stream) {
buffer += decoder.decode(chunk, { stream: true });
const parsed = Bun.JSONL.parseChunk(buffer);
for (const value of parsed.values) handleRecord(value);
buffer = buffer.slice(parsed.read);
}
for (const value of Bun.JSONL.parse(`${buffer}\n`)) handleRecord(value);
flushRequest();
return scan;
}
function sizeBucket(changedLines: number): string {
for (const [label, max] of SIZE_BUCKETS) {
if (changedLines <= max) return label;
}
return SIZE_BUCKETS[SIZE_BUCKETS.length - 1][0];
}
function printDistribution(title: string, counts: Map<string, number>, order?: string[]): void {
const total = [...counts.values()].reduce((a, b) => a + b, 0);
const keys = order ?? [...counts.keys()].sort();
console.log(title);
for (const key of keys) {
const count = counts.get(key) ?? 0;
console.log(` ${key.padEnd(6)} ${String(count).padStart(6)} ${((count / total) * 100).toFixed(1)}%`);
}
}
async function main(): Promise<number> {
const args = parseArguments();
const sessionFiles = (await fs.readdir(args.sessionsDir))
.filter(name => name.endsWith(".jsonl"))
.sort()
.reverse()
.slice(0, args.maxSessions > 0 ? args.maxSessions : undefined)
.map(name => path.join(args.sessionsDir, name));
console.log(`Scanning ${sessionFiles.length} session files in ${args.sessionsDir}`);
interface ShapeCounters {
sizes: Map<string, number>;
hunks: Map<string, number>;
ops: Map<string, number>;
changed: number[];
}
const newCounters = (): ShapeCounters => ({ sizes: new Map(), hunks: new Map(), ops: new Map(), changed: [] });
const accumulate = (counters: ShapeCounters, runs: ChangeRun[]): void => {
const changed = runs.reduce((sum, run) => sum + Math.max(run.oldCount, run.newCount), 0);
counters.changed.push(changed);
counters.sizes.set(sizeBucket(changed), (counters.sizes.get(sizeBucket(changed)) ?? 0) + 1);
const hunkKey = runs.length >= 3 ? "3+" : String(runs.length);
counters.hunks.set(hunkKey, (counters.hunks.get(hunkKey) ?? 0) + 1);
for (const run of runs) {
const op = run.oldCount === 0 ? "insert" : run.newCount === 0 ? "delete" : "replace";
counters.ops.set(op, (counters.ops.get(op) ?? 0) + 1);
}
};
const report = (label: string, counters: ShapeCounters): void => {
counters.changed.sort((a, b) => a - b);
const median = counters.changed[Math.floor(counters.changed.length / 2)] ?? 0;
console.log(`\n=== ${label}: ${counters.changed.length} samples, median changed lines: ${median}`);
printDistribution(
"changed lines:",
counters.sizes,
SIZE_BUCKETS.map(([bucketLabel]) => bucketLabel),
);
printDistribution("hunks:", counters.hunks, ["1", "2", "3+"]);
printDistribution("op mix:", counters.ops, ["replace", "insert", "delete"]);
};
const perCall = newCounters();
const perRequest = newCounters();
const formatCounts = new Map<string, number>();
let skipped = 0;
let next = 0;
const workers = Array.from({ length: 16 }, async () => {
while (next < sessionFiles.length) {
const file = sessionFiles[next++];
const scan = await collectSessionEdits(file).catch(
(): SessionScan => ({ edits: [], requests: [], formats: new Map(), skipped: 0 }),
);
skipped += scan.skipped;
for (const [format, count] of scan.formats) {
formatCounts.set(format, (formatCounts.get(format) ?? 0) + count);
}
for (const runs of scan.edits) accumulate(perCall, runs);
for (const runs of scan.requests) accumulate(perRequest, runs);
}
});
await Promise.all(workers);
if (perCall.changed.length === 0) {
console.error("No parseable edits found.");
return 1;
}
const parsedTotal = [...formatCounts.values()].reduce((a, b) => a + b, 0);
const coverage = parsedTotal + skipped > 0 ? parsedTotal / (parsedTotal + skipped) : 0;
const formatSummary = [...formatCounts.entries()].map(([format, count]) => `${format}=${count}`).join(", ");
console.log(
`\nParse coverage: ${parsedTotal}/${parsedTotal + skipped} diff results (${(coverage * 100).toFixed(1)}%; ${formatSummary})`,
);
report("per user request (calibration target)", perRequest);
report("per tool call", perCall);
return 0;
}
process.exit(await main());
@@ -12,20 +12,38 @@
* - Dense code: Minimal whitespace makes context harder to read
* - Deep nesting: Whitespace-sensitive edits at high indent levels
*
* Difficulty modes control both FILE SELECTION and PROMPT DETAIL:
* - easy: Short files, unique lines, line number given
* - medium: Medium files, function context given
* Difficulty modes control both FILE SELECTION and PROMPT DETAIL. Every prompt
* fully determines the byte-exact fix (before/after blocks, validated by
* re-solving them against the expected output); tiers vary how much locating
* the model must do:
* - easy: Short files, unique lines, exact line number given
* - medium: Medium files, containing function given
* - hard: Long files with similar blocks, no location hint
* - nightmare: Long files where target line repeats, minimal info
* - nightmare: Long files where nearly identical regions repeat, no location hint
*/
import * as fs from "node:fs";
import * as path from "node:path";
import { parseArgs } from "node:util";
import { TempDir } from "@oh-my-pi/pi-utils";
import { prompt, TempDir } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
import { diffLines } from "diff";
import { formatContent } from "./formatter";
import {
commonPrefixLength,
commonSuffixLength,
LANGUAGE_BY_EXTENSION,
mergeNearbyPlacements,
type Placement,
pickFence,
placementsFromDiff,
type RenderedHunk,
renderHunks,
solveRenderedHunks,
} from "./hunks";
import { ALL_MUTATIONS, CATEGORY_MAP, type Mutation, type MutationInfo } from "./mutations";
import identifierTaskTemplate from "./prompts/identifier-task.md" with { type: "text" };
import mutationTaskTemplate from "./prompts/mutation-task.md" with { type: "text" };
import structuralTaskTemplate from "./prompts/structural-task.md" with { type: "text" };
const SUPPORTED_EXTENSIONS = new Set([".js", ".jsx", ".ts", ".tsx"]);
using DEFAULT_SOURCE_REPO_DIR = TempDir.createSync("@pi-mono-source");
@@ -64,6 +82,7 @@ interface CaseResult {
filePath: string;
formattedMutatedContent: string;
formattedOriginalContent: string;
prompt: string;
difficulty: Difficulty;
difficultyScore: number;
}
@@ -71,7 +90,7 @@ interface CaseResult {
interface Args {
typescriptDir: string;
output: string;
countPerType: number;
countScale: number;
seed: number;
categories: string | null;
difficulty: string;
@@ -84,7 +103,7 @@ function parseArguments(): Args {
options: {
"typescript-dir": { type: "string", default: DEFAULT_SOURCE_REPO_DIR.path() },
output: { type: "string", default: DEFAULT_OUTPUT },
"count-per-type": { type: "string", default: "20" },
"count-scale": { type: "string", default: "1" },
seed: { type: "string", default: "42" },
categories: { type: "string" },
difficulty: { type: "string", default: "easy,medium,hard,nightmare" },
@@ -96,7 +115,7 @@ function parseArguments(): Args {
return {
typescriptDir: values["typescript-dir"] ?? DEFAULT_SOURCE_REPO_DIR.path(),
output: values.output ?? DEFAULT_OUTPUT,
countPerType: parseInt(values["count-per-type"] ?? "20", 10),
countScale: Number.parseFloat(values["count-scale"] ?? "1"),
seed: parseInt(values.seed ?? "42", 10),
categories: values.categories ?? null,
difficulty: values.difficulty ?? "easy,medium,hard,nightmare",
@@ -105,6 +124,59 @@ function parseArguments(): Args {
};
}
/** Target changed-line range for a generated case. */
type SizeRange = [number, number];
/**
* Per-mutation case counts and changed-line targets, calibrated against the
* empirical edit-shape distribution measured from real agent sessions
* (`edit-shape-stats.ts`): 1 → 27%, 2-5 → 30%, 6-20 → 26%, 21-60 → 13%,
* 61+ → 4%. Token-level mutations supply the 1-line mass; small structural and
* multi-site mutations the 2-5 band; block-level mutations cycle through the
* larger bands via `sizes`.
*/
interface MutationPlan {
count: number;
sizes?: SizeRange[];
}
const BLOCK_SIZES: SizeRange[] = [
[6, 20],
[6, 20],
[21, 60],
[6, 20],
[21, 60],
[6, 20],
[61, 150],
[21, 60],
];
const MUTATION_PLANS: Record<string, MutationPlan> = {
"swap-comparison": { count: 2 },
"swap-equality": { count: 2 },
"swap-logical": { count: 2 },
"remove-negation": { count: 2 },
"swap-increment-decrement": { count: 2 },
"swap-arithmetic": { count: 2 },
"flip-boolean": { count: 2 },
"remove-optional-chain": { count: 2 },
"swap-call-args": { count: 2 },
"swap-nullish": { count: 2 },
"swap-regex-quantifier": { count: 2 },
"unicode-hyphen": { count: 2 },
"off-by-one": { count: 2 },
"swap-adjacent-lines": { count: 6 },
"duplicate-line-flip": { count: 6 },
"identifier-multi-edit": { count: 8 },
"swap-if-else": { count: 6 },
"wrap-redundant-if": { count: 12, sizes: BLOCK_SIZES },
"swap-sibling-blocks": { count: 12, sizes: BLOCK_SIZES },
"duplicate-block": { count: 6, sizes: BLOCK_SIZES },
"move-distant-block": { count: 14, sizes: BLOCK_SIZES },
"remove-case-label": { count: 8 },
"composite-multi-edit": { count: 14 },
};
async function ensureSourceRepo(typescriptDir: string): Promise<void> {
if (fs.existsSync(typescriptDir)) {
const packagesDir = path.join(typescriptDir, "packages");
@@ -378,69 +450,295 @@ function recordRegion(usedLines: Map<string, number[]>, filePath: string, lineNu
usedLines.set(filePath, used);
}
function escapeRegExp(value: string): string {
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
/** Where a bug sits for medium prompts: containing function, else file-third region. */
function locateBug(
content: string,
lineNumber: number,
lineCount: number,
): { functionName?: string; region?: "top" | "middle" | "end" } {
for (const [name, start, end] of findFunctionRanges(content)) {
if (lineNumber >= start && lineNumber <= end) return { functionName: name };
}
const ratio = lineCount > 0 ? lineNumber / lineCount : 0;
return { region: ratio < 0.33 ? "top" : ratio < 0.66 ? "middle" : "end" };
}
/**
* Rename-style prompt for identifier-multi-edit: names the misspelled and
* correct identifiers so "replace every occurrence" admits exactly one
* byte-exact answer. Returns null when a plain word-boundary rename does not
* reproduce the expected file (e.g. the misspelling collides with other text).
*/
function buildIdentifierPrompt(
filePath: string,
info: MutationInfo,
difficulty: Difficulty,
input: string,
expected: string,
): string | null {
const pair = info.identifier;
if (!pair) return null;
const wordRe = new RegExp(`(?<![A-Za-z0-9_$])${escapeRegExp(pair.misspelled)}(?![A-Za-z0-9_$])`, "g");
if (input.replace(wordRe, pair.correct) !== expected) return null;
const affectedLines: number[] = [];
input.split("\n").forEach((line, index) => {
if (wordRe.test(line)) affectedLines.push(index + 1);
wordRe.lastIndex = 0;
});
if (affectedLines.length === 0) return null;
return prompt.render(identifierTaskTemplate, {
filename: path.basename(filePath),
correct: pair.correct,
misspelled: pair.misspelled,
count: affectedLines.length,
affectedLines: difficulty === "easy" ? affectedLines : undefined,
});
}
/** Mutations whose fixtures read as natural instructions instead of raw before/after patches. */
type StructuralKind =
| "case-label"
| "duplicate-block"
| "move-block"
| "wrap-if"
| "swap-blocks"
| "swap-lines"
| "swap-if-else";
const STRUCTURAL_KINDS: Record<string, StructuralKind> = {
"remove-case-label": "case-label",
"duplicate-block": "duplicate-block",
"move-distant-block": "move-block",
"wrap-redundant-if": "wrap-if",
"swap-sibling-blocks": "swap-blocks",
"swap-adjacent-lines": "swap-lines",
"swap-if-else": "swap-if-else",
};
function countTrimmedLine(lines: string[], needle: string): number {
let count = 0;
for (const line of lines) {
if (line.trim() === needle) count++;
}
return count;
}
/** Whether a line is a usable prose anchor (contains an identifier, not just punctuation like `);`). */
function isAnchorLine(line: string): boolean {
return /[A-Za-z_$]{2,}/.test(line);
}
function firstNonEmptyTrimmed(lines: string[]): string {
for (const line of lines) {
const trimmed = line.trim();
if (trimmed) return trimmed;
}
return "";
}
/** A placement reduced to its changed core (shared context trimmed away). */
function trimPlacement(
inputLines: string[],
placement: Placement,
): { start: number; oldCore: string[]; newCore: string[] } {
const old = inputLines.slice(placement.start, placement.start + placement.oldLen);
const fresh = placement.newLines;
const prefix = commonPrefixLength(old, fresh);
const suffix = commonSuffixLength(old, fresh, Math.min(old.length, fresh.length) - prefix);
return {
start: placement.start + prefix,
oldCore: old.slice(prefix, old.length - suffix),
newCore: fresh.slice(prefix, fresh.length - suffix),
};
}
/** Nearest non-blank line trimmed, scanning from `index` in `direction`. */
function nearestVisibleLine(lines: string[], index: number, direction: 1 | -1): string {
for (let i = index; i >= 0 && i < lines.length; i += direction) {
const trimmed = lines[i].trim();
if (trimmed) return trimmed;
}
return "";
}
/**
* Derive instruction data for a structural mutation from its placements.
* Every referenced anchor line is validated to occur exactly the expected
* number of times, so the instruction identifies a unique edit; returns null
* when the instruction would be ambiguous (the caller then rejects the
* candidate and retries — structural kinds never ship raw-patch fallbacks).
*/
function structuralPromptData(
kind: StructuralKind,
inputLines: string[],
placements: Placement[],
): Record<string, unknown> | null {
const cores = placements.map(placement => trimPlacement(inputLines, placement));
const visible = (lines: string[]): string[] => lines.filter(line => line.trim() !== "");
switch (kind) {
case "case-label": {
if (cores.length !== 1) return null;
const { start, oldCore, newCore } = cores[0];
const added = visible(newCore);
if (visible(oldCore).length !== 0 || added.length !== 1) return null;
const label = added[0].trim();
const before = nearestVisibleLine(inputLines, start + oldCore.length, 1);
if (!label.startsWith("case") || !(before.startsWith("case") || before.startsWith("default"))) return null;
if (countTrimmedLine(inputLines, before) !== 1) return null;
return { label, before };
}
case "duplicate-block": {
if (cores.length !== 1) return null;
const { oldCore, newCore } = cores[0];
if (visible(newCore).length !== 0 || visible(oldCore).length === 0) return null;
const head = firstNonEmptyTrimmed(oldCore);
if (!head || countTrimmedLine(inputLines, head) !== 2) return null;
return { head };
}
case "move-block": {
if (cores.length !== 2) return null;
const insert = cores.find(core => visible(core.oldCore).length === 0);
const remove = cores.find(core => visible(core.newCore).length === 0);
if (!insert || !remove || insert === remove) return null;
if (visible(remove.oldCore).join("\n") !== visible(insert.newCore).join("\n")) return null;
const head = firstNonEmptyTrimmed(remove.oldCore);
const destination = nearestVisibleLine(inputLines, insert.start + insert.oldCore.length, 1);
const currentPrev = nearestVisibleLine(inputLines, remove.start - 1, -1);
if (![head, destination, currentPrev].every(isAnchorLine)) return null;
for (const anchor of [head, destination, currentPrev]) {
if (countTrimmedLine(inputLines, anchor) !== 1) return null;
}
return { head, destination, currentPrev };
}
case "wrap-if": {
if (cores.length !== 1) return null;
const { start, oldCore } = cores[0];
const relative = oldCore.findIndex(line => line.trim() === "if (true) {");
if (relative < 0) return null;
return { wrapperLine: start + relative + 1 };
}
case "swap-blocks":
case "swap-lines": {
const merged = mergeNearbyPlacements(inputLines, placements, 2);
let firstHead = "";
let secondHead = "";
if (merged.length === 1) {
const { oldCore, newCore } = trimPlacement(inputLines, merged[0]);
firstHead = firstNonEmptyTrimmed(oldCore);
secondHead = firstNonEmptyTrimmed(newCore);
} else if (cores.length === 2) {
// jsdiff renders `A,B -> B,A` as insert-A-before-B plus delete-A-after-B,
// with all of B as unchanged context in between; match the moved body.
const insert = cores.find(core => visible(core.oldCore).length === 0);
const remove = cores.find(core => visible(core.newCore).length === 0);
if (!insert || !remove || insert === remove) return null;
if (visible(remove.oldCore).join("\n") !== visible(insert.newCore).join("\n")) return null;
secondHead = firstNonEmptyTrimmed(insert.newCore);
firstHead = nearestVisibleLine(inputLines, insert.start + insert.oldCore.length, 1);
} else {
return null;
}
if (!firstHead || !secondHead || firstHead === secondHead) return null;
if (!isAnchorLine(firstHead) || !isAnchorLine(secondHead)) return null;
if (countTrimmedLine(inputLines, firstHead) !== 1 || countTrimmedLine(inputLines, secondHead) !== 1)
return null;
return { firstHead, secondHead };
}
case "swap-if-else": {
if (cores.length > 2) return null;
const condition = nearestVisibleLine(inputLines, cores[0].start - 1, -1);
if (!condition.startsWith("if") && !condition.includes(" if ")) return null;
if (countTrimmedLine(inputLines, condition) !== 1) return null;
return { condition };
}
}
}
/**
* Build the task prompt for a mutation case, or null when no uniquely solvable
* prompt exists (the caller then retries with a different candidate).
*
* Structural mutations render as natural instructions ("move X back before Y")
* with the expected after-state as reference; identifier renames as a
* replace-every-occurrence instruction; everything else as explicit
* before/after blocks. All variants are validated by re-solving the hunks
* against the expected output, so every prompt admits exactly one byte-exact
* answer. Difficulty only controls location detail.
*/
function buildPrompt(
filePath: string,
mutation: Mutation,
info: MutationInfo,
difficulty: Difficulty,
entry: FileEntry,
): string {
const header = `# Fix the bug in \`${path.basename(filePath)}\``;
const isStructural = mutation.category === "structural";
const isMultiEdit = mutation.name === "identifier-multi-edit";
if (difficulty === "easy") {
const detail = mutation.describe(info);
const location = isStructural
? `The issue starts around line ${info.lineNumber}.`
: `The issue is on line ${info.lineNumber}.`;
return [header, detail, location, mutation.fixHint].join("\n\n");
input: string,
expected: string,
): string | null {
if (mutation.name === "identifier-multi-edit") {
return buildIdentifierPrompt(filePath, info, difficulty, input, expected);
}
if (difficulty === "medium") {
const detail = mutation.describe(info);
const funcName = findContainingFunction(entry, info.lineNumber);
let location: string;
if (funcName) {
location = `The issue is in the \`${funcName}\` function.`;
} else {
const ratio = entry.lineCount ? info.lineNumber / entry.lineCount : 0;
if (ratio < 0.33) location = "The issue is near the top of the file.";
else if (ratio < 0.66) location = "The issue is around the middle of the file.";
else location = "The issue is near the end of the file.";
}
if (isMultiEdit) {
location += " The same error appears in multiple places.";
}
return [header, detail, location, mutation.fixHint].join("\n\n");
const inputLines = input.replace(/\n$/, "").split("\n");
const placements = placementsFromDiff(input, expected);
if (placements.length === 0 || placements.length > 5) return null;
const hunks = renderHunks(inputLines, placements);
if (!hunks) return null;
const solved = solveRenderedHunks(inputLines, hunks);
if (!solved || ensureTrailingNewline(solved.join("\n")) !== expected) return null;
const language = LANGUAGE_BY_EXTENSION[path.extname(filePath).toLowerCase()] ?? "";
const fence = pickFence(hunks);
const hunkContext = (hunk: RenderedHunk): { startLine: number | undefined } => ({
startLine: !hunk.unique || difficulty === "easy" ? hunk.startLine : undefined,
});
const structuralKind = STRUCTURAL_KINDS[mutation.name];
if (structuralKind) {
// Structural tasks read as natural instructions; a candidate whose
// instruction would be ambiguous is rejected so the caller retries —
// never downgraded to a raw before/after patch.
const data = structuralPromptData(structuralKind, inputLines, placements);
if (!data) return null;
const rendered = prompt.render(structuralTaskTemplate, {
filename: path.basename(filePath),
kind: structuralKind,
...data,
fence,
language,
hunkCount: hunks.length,
hunks: hunks.map(hunk => ({
// After-state regions are only interpretable with a position.
startLine: hunk.startLine,
newCode: hunk.newBlock.join("\n"),
})),
});
return rendered.length <= 20_000 ? rendered : null;
}
if (difficulty === "hard") {
const detail = mutation.describe(info);
if (isMultiEdit) {
return [header, detail, "Find and fix all occurrences of this issue."].join("\n\n");
}
if (isStructural) {
return [header, detail, "The fix may involve multiple lines."].join("\n\n");
}
return [header, detail, "Find and fix this issue."].join("\n\n");
}
// nightmare
if (isStructural) {
return [header, "There is a structural bug in this file.", "Track it down and fix it with a minimal edit."].join(
"\n\n",
);
}
if (isMultiEdit) {
return [
header,
"An identifier is consistently misspelled throughout this file.",
"Find all occurrences and fix them.",
].join("\n\n");
}
return [header, "There is a subtle bug in this file.", "Track it down and fix it with a minimal edit."].join("\n\n");
const location = difficulty === "medium" ? locateBug(input, hunks[0].startLine, inputLines.length) : {};
const rendered = prompt.render(mutationTaskTemplate, {
filename: path.basename(filePath),
name: mutation.name,
functionName: location.functionName,
region: location.region,
nightmare: difficulty === "nightmare",
fence,
language,
hunkCount: hunks.length,
hunks: hunks.map(hunk => ({
...hunkContext(hunk),
isDelete: hunk.newBlock.length === 0,
oldCode: hunk.oldBlock.join("\n"),
newCode: hunk.newBlock.join("\n"),
})),
});
return rendered.length <= 20_000 ? rendered : null;
}
function createSeededRng(seed: number): () => number {
@@ -625,6 +923,7 @@ function resolveFormattedMutationInfo(
lineNumber,
originalSnippet: snippetFromChangedLines(chosen.removedLines, originalLines, Math.max(1, chosen.oldStart)),
mutatedSnippet: snippetFromChangedLines(chosen.addedLines, mutatedLines, lineNumber),
identifier: fallbackInfo.identifier,
};
if (resolved.originalSnippet && !originalContent.includes(resolved.originalSnippet)) return null;
@@ -640,6 +939,7 @@ async function generateCase(
usedLines: Map<string, number[]>,
difficulty: Difficulty,
minScore: number | null,
sizeRange: SizeRange | null,
attemptLimit = 100,
): Promise<CaseResult | null> {
let candidates = getCandidatesForDifficulty(files, difficulty);
@@ -676,12 +976,15 @@ async function generateCase(
const positionalHunks = countPositionalLineHunks(entry.content, mutatedContent);
const positionalChanges = countPositionalLineChanges(entry.content, mutatedContent);
const isStructural = mutation.category === "structural";
const isMultiEdit = mutation.name === "identifier-multi-edit";
const isMultiEdit = mutation.multiHunk === true;
if (!isStructural && !isMultiEdit) {
if (changedHunks !== 1) continue;
if (positionalHunks !== 1) continue;
if (changedLines > 30 || positionalChanges > 30) continue;
}
// Coarse pre-format lower bound: removed+added always >= the run-based
// metric, so anything below the minimum can be rejected before Prettier.
if (sizeRange && changedLines < sizeRange[0]) continue;
if (!regionAvailable(usedLines, entry.path, rawInfo.lineNumber)) continue;
@@ -707,6 +1010,26 @@ async function generateCase(
const finalInfo = resolveFormattedMutationInfo(normalizedOriginal, normalizedMutated, rawInfo);
if (!finalInfo) continue;
// Enforce the size target with the same metric edit-shape-stats.ts uses:
// sum of max(old, new) per change run, on the formatted input/expected.
if (sizeRange) {
const finalChanged = placementsFromDiff(normalizedMutated, normalizedOriginal).reduce(
(sum, placement) => sum + Math.max(placement.oldLen, placement.newLines.length),
0,
);
if (finalChanged < sizeRange[0] || finalChanged > sizeRange[1]) continue;
}
const taskPrompt = buildPrompt(
entry.path,
mutation,
finalInfo,
difficulty,
normalizedMutated,
normalizedOriginal,
);
if (!taskPrompt) continue;
recordRegion(usedLines, entry.path, rawInfo.lineNumber);
return {
caseId: "",
@@ -715,6 +1038,7 @@ async function generateCase(
filePath: entry.path,
formattedMutatedContent: normalizedMutated,
formattedOriginalContent: normalizedOriginal,
prompt: taskPrompt,
difficulty,
difficultyScore: diffScore,
};
@@ -760,7 +1084,6 @@ async function buildCaseEntries(result: CaseResult, typescriptDir: string): Prom
};
const isRepeated = entry.repeatedLines.has(lineContent);
const prompt = buildPrompt(result.filePath, result.mutation, result.finalInfo, result.difficulty, entry);
const metadata = {
mutation_type: result.mutation.name,
@@ -796,7 +1119,7 @@ async function buildCaseEntries(result: CaseResult, typescriptDir: string): Prom
return [
{ name: `${caseDir}/input/${filename}`, content: result.formattedMutatedContent },
{ name: `${caseDir}/expected/${filename}`, content: result.formattedOriginalContent },
{ name: `${caseDir}/prompt.md`, content: prompt },
{ name: `${caseDir}/prompt.md`, content: result.prompt },
{ name: `${caseDir}/metadata.json`, content: JSON.stringify(metadata, null, 2) },
];
}
@@ -856,17 +1179,20 @@ async function main(): Promise<number> {
const fallbackOrder: Difficulty[] = ["hard", "medium", "easy"];
for (const mutation of mutations) {
const difficultiesForType = chooseDifficulties(difficulties, args.countPerType);
const plan = MUTATION_PLANS[mutation.name] ?? { count: 4 };
const count = Math.max(1, Math.round(plan.count * args.countScale));
const difficultiesForType = chooseDifficulties(difficulties, count);
let generated = 0;
for (let index = 0; index < args.countPerType; index++) {
for (let index = 0; index < count; index++) {
const difficulty = difficultiesForType[index];
let result = await generateCase(rng, mutation, files, usedLines, difficulty, args.minScore);
const sizeRange = plan.sizes ? plan.sizes[index % plan.sizes.length] : null;
let result = await generateCase(rng, mutation, files, usedLines, difficulty, args.minScore, sizeRange);
if (!result) {
for (const fallback of fallbackOrder) {
if (fallback === difficulty) continue;
result = await generateCase(rng, mutation, files, usedLines, fallback, 0);
result = await generateCase(rng, mutation, files, usedLines, fallback, 0, sizeRange);
if (result) {
console.log(`Note: ${mutation.name} case ${index + 1} fell back from ${difficulty} to ${fallback}`);
break;
@@ -889,8 +1215,8 @@ async function main(): Promise<number> {
if (generated === 0) {
console.log(`Warning: No cases generated for ${mutation.name} (mutation may be too rare)`);
} else if (generated < args.countPerType) {
console.log(`Note: Only ${generated}/${args.countPerType} cases generated for ${mutation.name}`);
} else if (generated < count) {
console.log(`Note: Only ${generated}/${count} cases generated for ${mutation.name}`);
}
}
@@ -0,0 +1,264 @@
/**
* Shared hunk machinery for fixture generation.
*
* `generate.ts` expresses every task as "replace these exact blocks in the
* input file". This module locates changed blocks, prepares prompt-visible
* before/after blocks whose old side occurs exactly once in the input
* (re-padding with context, or falling back to an explicit line number), and
* re-solves prepared hunks to prove a prompt admits exactly one byte-exact
* answer. Prompt prose itself lives in the Handlebars templates under
* `src/prompts/`; this module only supplies the block data.
*/
import { diffLines } from "diff";
/** Replace `oldLen` lines at `start` (0-based index into the input) with `newLines`. */
export interface Placement {
start: number;
oldLen: number;
newLines: string[];
}
/** A prompt-visible change: replace (or delete) `oldBlock`, verbatim, with `newBlock`. */
export interface RenderedHunk {
oldBlock: string[];
newBlock: string[];
/** 1-based line where `oldBlock` starts in the input file. */
startLine: number;
/** Whether `oldBlock` occurs exactly once in the input; when false, prompts must state `startLine`. */
unique: boolean;
}
const MAX_UNIQUENESS_EXTENSION = 10;
export function findBlockOccurrences(lines: string[], block: string[]): number[] {
if (block.length === 0 || block.length > lines.length) return [];
const hits: number[] = [];
const first = block[0];
outer: for (let i = 0; i <= lines.length - block.length; i++) {
if (lines[i] !== first) continue;
for (let j = 1; j < block.length; j++) {
if (lines[i + j] !== block[j]) continue outer;
}
hits.push(i);
}
return hits;
}
export function applyPlacements(lines: string[], placements: Placement[]): string[] {
const out: string[] = [];
let cursor = 0;
for (const placement of placements) {
out.push(...lines.slice(cursor, placement.start), ...placement.newLines);
cursor = placement.start + placement.oldLen;
}
out.push(...lines.slice(cursor));
return out;
}
export function commonPrefixLength(a: string[], b: string[]): number {
let n = 0;
while (n < a.length && n < b.length && a[n] === b[n]) n++;
return n;
}
export function commonSuffixLength(a: string[], b: string[], maxLength: number): number {
let n = 0;
while (n < maxLength && a[a.length - 1 - n] === b[b.length - 1 - n]) n++;
return n;
}
function splitDiffValue(value: string): string[] {
const lines = value.split("\n");
if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
return lines;
}
/**
* Compute placements that transform `inputText` into `expectedText` via a
* line diff. Adjacent removed/added runs merge into one placement.
*/
export function placementsFromDiff(inputText: string, expectedText: string): Placement[] {
const placements: Placement[] = [];
let inputLine = 0;
let pendingStart = -1;
let pendingOldLen = 0;
let pendingNewLines: string[] = [];
const flush = (): void => {
if (pendingStart < 0) return;
placements.push({ start: pendingStart, oldLen: pendingOldLen, newLines: pendingNewLines });
pendingStart = -1;
pendingOldLen = 0;
pendingNewLines = [];
};
for (const change of diffLines(inputText, expectedText)) {
const lines = splitDiffValue(change.value);
if (!change.added && !change.removed) {
flush();
inputLine += lines.length;
continue;
}
if (pendingStart < 0) pendingStart = inputLine;
if (change.removed) {
pendingOldLen += lines.length;
inputLine += lines.length;
} else {
pendingNewLines.push(...lines);
}
}
flush();
return placements;
}
/**
* Merge placements separated by at most `maxGap` context lines into one, so a
* moved statement renders as a single natural replace instead of a delete plus
* a re-insert whose blocks end on invisible blank lines.
*/
export function mergeNearbyPlacements(inputLines: string[], placements: Placement[], maxGap: number): Placement[] {
const merged: Placement[] = [];
for (const placement of placements) {
const previous = merged[merged.length - 1];
const gap = previous ? placement.start - (previous.start + previous.oldLen) : -1;
if (previous && gap >= 0 && gap <= maxGap) {
const bridge = inputLines.slice(previous.start + previous.oldLen, placement.start);
previous.oldLen += gap + placement.oldLen;
previous.newLines = [...previous.newLines, ...bridge, ...placement.newLines];
continue;
}
merged.push({ ...placement });
}
return merged;
}
/**
* Turn placements into prompt-visible hunks. Each hunk's blocks are trimmed to
* the changed core, anchored on context when either side would be empty, then
* re-padded with context lines until the old block occurs exactly once in the
* input; when the input has no room to disambiguate, the hunk is marked
* non-unique and prompts must state its start line. Blocks never start or end
* on a blank line when a visible context line is available — a blank line at a
* fence edge is invisible. Returns null when a hunk cannot be rendered at all.
*/
export function renderHunks(inputLines: string[], placements: Placement[]): RenderedHunk[] | null {
const merged = mergeNearbyPlacements(inputLines, placements, 2);
const hunks: RenderedHunk[] = [];
let previousEnd = -1;
for (let index = 0; index < merged.length; index++) {
const placement = merged[index];
const nextStart = index + 1 < merged.length ? merged[index + 1].start : inputLines.length;
const old = inputLines.slice(placement.start, placement.start + placement.oldLen);
const fresh = placement.newLines;
const prefix = commonPrefixLength(old, fresh);
const suffix = commonSuffixLength(old, fresh, Math.min(old.length, fresh.length) - prefix);
let shownStart = placement.start + prefix;
let oldBlock = old.slice(prefix, old.length - suffix);
let newBlock = fresh.slice(prefix, fresh.length - suffix);
if (oldBlock.length === 0 && newBlock.length === 0) continue;
const takeAbove = (): boolean => {
const above = shownStart - 1;
if (above < 0 || above <= previousEnd) return false;
shownStart = above;
oldBlock = [inputLines[above], ...oldBlock];
newBlock = [inputLines[above], ...newBlock];
return true;
};
const takeBelow = (): boolean => {
const below = shownStart + oldBlock.length;
if (below >= nextStart || below >= inputLines.length) return false;
oldBlock = [...oldBlock, inputLines[below]];
newBlock = [...newBlock, inputLines[below]];
return true;
};
// Pure insertion or pure deletion: anchor on a context line so neither
// block is empty. Prefer a non-blank anchor.
if (oldBlock.length === 0 || newBlock.length === 0) {
const aboveIndex = shownStart - 1;
const aboveVisible = aboveIndex >= 0 && aboveIndex > previousEnd && inputLines[aboveIndex].trim() !== "";
if (aboveVisible) {
takeAbove();
} else if (!takeBelow() && !takeAbove()) {
return null;
}
}
let unique = true;
for (let extension = 0; findBlockOccurrences(inputLines, oldBlock).length > 1; extension++) {
if (extension >= MAX_UNIQUENESS_EXTENSION || !takeAbove()) {
unique = false;
break;
}
}
// Pad blank fence edges with visible context (bounded; best effort).
for (let guard = 0; guard < 2 && (oldBlock[0].trim() === "" || newBlock[0]?.trim() === ""); guard++) {
if (!takeAbove()) break;
}
for (
let guard = 0;
guard < 2 && (oldBlock[oldBlock.length - 1].trim() === "" || newBlock[newBlock.length - 1]?.trim() === "");
guard++
) {
if (!takeBelow()) break;
}
previousEnd = placement.start + placement.oldLen - 1;
hunks.push({ oldBlock, newBlock, startLine: shownStart + 1, unique });
}
return hunks.length > 0 ? hunks : null;
}
/**
* Re-solve the prompt from scratch: locate every hunk's old block in the input
* (unique occurrence, or the stated start line) and apply the replacements.
* Guarantees the prompt admits exactly one byte-exact answer.
*/
export function solveRenderedHunks(inputLines: string[], hunks: RenderedHunk[]): string[] | null {
const placements: Placement[] = [];
for (const hunk of hunks) {
const occurrences = findBlockOccurrences(inputLines, hunk.oldBlock);
let start: number;
if (hunk.unique) {
if (occurrences.length !== 1) return null;
start = occurrences[0];
} else {
start = hunk.startLine - 1;
if (!occurrences.includes(start)) return null;
}
placements.push({ start, oldLen: hunk.oldBlock.length, newLines: hunk.newBlock });
}
placements.sort((a, b) => a.start - b.start);
let previousEnd = -1;
for (const placement of placements) {
if (placement.start <= previousEnd) return null;
previousEnd = placement.start + placement.oldLen - 1;
}
return applyPlacements(inputLines, placements);
}
export const LANGUAGE_BY_EXTENSION: Record<string, string> = {
".ts": "ts",
".tsx": "tsx",
".js": "js",
".jsx": "jsx",
".mjs": "js",
".md": "markdown",
".rs": "rust",
".py": "python",
".json": "json",
".yml": "yaml",
".yaml": "yaml",
".toml": "toml",
};
/** Fence that safely wraps every block in `hunks` (upgrades when content embeds triple backticks). */
export function pickFence(hunks: RenderedHunk[]): string {
const hasTripleBacktick = hunks.some(hunk =>
[...hunk.oldBlock, ...hunk.newBlock].some(line => line.includes("```")),
);
return hasTripleBacktick ? "````" : "```";
}
@@ -21,16 +21,18 @@ export interface MutationInfo {
lineNumber: number;
originalSnippet: string;
mutatedSnippet: string;
/** Correct/misspelled rename pair; set by identifier-multi-edit for rename-style prompts. */
identifier?: { correct: string; misspelled: string };
}
export interface Mutation {
name: string;
category: string;
fixHint: string;
/** Whether this mutation intentionally produces multiple separated hunks. */
multiHunk?: boolean;
canApply(content: string): boolean;
mutate(content: string, rng: () => number): [string, MutationInfo];
describe(info: MutationInfo): string;
}
type Candidate<TNode extends t.Node = t.Node, TMeta = unknown> = {
@@ -234,12 +236,6 @@ function isLengthMemberExpression(node: t.Node): node is t.MemberExpression {
abstract class BaseAstMutation implements Mutation {
abstract name: string;
abstract category: string;
abstract fixHint: string;
abstract description: string;
describe(_info: MutationInfo): string {
return this.description;
}
abstract collectCandidates(parsed: Parsed): Candidate[];
abstract applyCandidate(parsed: Parsed, candidate: Candidate, rng: () => number): MutationInfo;
@@ -281,8 +277,6 @@ abstract class BaseAstMutation implements Mutation {
class SwapComparisonMutation extends BaseAstMutation {
name = "swap-comparison";
category = "operator";
fixHint = "Swap the comparison operator to the correct variant.";
description = "A comparison operator is subtly wrong.";
#swap: Record<string, t.BinaryExpression["operator"]> = {
"<=": "<",
@@ -310,8 +304,6 @@ class SwapComparisonMutation extends BaseAstMutation {
class SwapEqualityMutation extends BaseAstMutation {
name = "swap-equality";
category = "operator";
fixHint = "Fix the equality comparison operator.";
description = "An equality operator is inverted.";
#swap: Record<string, t.BinaryExpression["operator"]> = {
"===": "!==",
@@ -339,8 +331,6 @@ class SwapEqualityMutation extends BaseAstMutation {
class SwapLogicalMutation extends BaseAstMutation {
name = "swap-logical";
category = "operator";
fixHint = "Use the intended boolean operator.";
description = "A boolean operator is incorrect.";
collectCandidates(parsed: Parsed): Candidate<t.LogicalExpression>[] {
const out: Candidate<t.LogicalExpression>[] = [];
@@ -364,8 +354,6 @@ class SwapLogicalMutation extends BaseAstMutation {
class RemoveNegationMutation extends BaseAstMutation {
name = "remove-negation";
category = "operator";
fixHint = "Add back the missing logical negation (`!`).";
description = "A logical negation (`!`) was accidentally removed.";
collectCandidates(parsed: Parsed): Candidate<t.UnaryExpression>[] {
const out: Candidate<t.UnaryExpression>[] = [];
@@ -393,8 +381,6 @@ class RemoveNegationMutation extends BaseAstMutation {
class SwapIncDecMutation extends BaseAstMutation {
name = "swap-increment-decrement";
category = "operator";
fixHint = "Replace the increment/decrement operator with the intended one.";
description = "An increment/decrement operator points the wrong direction.";
collectCandidates(parsed: Parsed): Candidate<t.UpdateExpression>[] {
const out: Candidate<t.UpdateExpression>[] = [];
@@ -417,8 +403,6 @@ class SwapIncDecMutation extends BaseAstMutation {
class SwapArithmeticMutation extends BaseAstMutation {
name = "swap-arithmetic";
category = "operator";
fixHint = "Correct the arithmetic operator.";
description = "An arithmetic operator was swapped.";
#swap: Record<string, t.BinaryExpression["operator"]> = { "+": "-", "-": "+", "*": "/", "/": "*" };
@@ -441,8 +425,6 @@ class SwapArithmeticMutation extends BaseAstMutation {
class BooleanLiteralFlipMutation extends BaseAstMutation {
name = "flip-boolean";
category = "literal";
fixHint = "Flip the boolean literal to the intended value.";
description = "A boolean literal is inverted.";
collectCandidates(parsed: Parsed): Candidate<t.BooleanLiteral>[] {
const out: Candidate<t.BooleanLiteral>[] = [];
@@ -465,9 +447,6 @@ class BooleanLiteralFlipMutation extends BaseAstMutation {
class OptionalChainRemovalMutation extends BaseAstMutation {
name = "remove-optional-chain";
category = "access";
fixHint =
"Restore the optional chaining operator (`?.`) at the ONE location where it was removed. Do not add optional chaining elsewhere.";
description = "Optional chaining was removed from a property access.";
collectCandidates(parsed: Parsed): Candidate<t.OptionalMemberExpression | t.OptionalCallExpression>[] {
const out: Candidate<t.OptionalMemberExpression | t.OptionalCallExpression>[] = [];
@@ -498,8 +477,6 @@ class OptionalChainRemovalMutation extends BaseAstMutation {
class CallArgumentSwapMutation extends BaseAstMutation {
name = "swap-call-args";
category = "call";
fixHint = "Swap the two arguments to their original order.";
description = "Two arguments in a call are swapped.";
collectCandidates(parsed: Parsed): Candidate<t.CallExpression>[] {
const out: Candidate<t.CallExpression>[] = [];
@@ -563,8 +540,6 @@ class CallArgumentSwapMutation extends BaseAstMutation {
class NullishCoalescingSwapMutation extends BaseAstMutation {
name = "swap-nullish";
category = "operator";
fixHint = "Use the intended nullish/logical operator.";
description = "A nullish coalescing operator was swapped.";
collectCandidates(parsed: Parsed): Candidate<t.LogicalExpression>[] {
const out: Candidate<t.LogicalExpression>[] = [];
@@ -588,8 +563,6 @@ class NullishCoalescingSwapMutation extends BaseAstMutation {
class RegexQuantifierSwapMutation extends BaseAstMutation {
name = "swap-regex-quantifier";
category = "regex";
fixHint = "Fix the ONE regex quantifier that was swapped (between `+` and `*`). Do not modify other quantifiers.";
description = "A regex quantifier was swapped, changing whitespace matching.";
collectCandidates(parsed: Parsed): Candidate<t.RegExpLiteral>[] {
const out: Candidate<t.RegExpLiteral>[] = [];
@@ -650,8 +623,6 @@ class RegexQuantifierSwapMutation extends BaseAstMutation {
class UnicodeHyphenMutation extends BaseAstMutation {
name = "unicode-hyphen";
category = "unicode";
fixHint = "Replace the unicode dash with a plain ASCII hyphen.";
description = "A string literal contains a lookalike unicode dash.";
collectCandidates(parsed: Parsed): Candidate<t.StringLiteral | t.TemplateElement>[] {
const out: Candidate<t.StringLiteral | t.TemplateElement>[] = [];
@@ -690,8 +661,7 @@ class UnicodeHyphenMutation extends BaseAstMutation {
class IdentifierMultiEditMutation extends BaseAstMutation {
name = "identifier-multi-edit";
category = "identifier";
fixHint = "Restore the identifier to its original spelling in all affected locations.";
description = "An identifier is misspelled in multiple separate locations.";
multiHunk = true;
#keywords = new Set([
"await",
@@ -850,6 +820,7 @@ class IdentifierMultiEditMutation extends BaseAstMutation {
lineNumber: selectedPaths[0]?.node.loc?.start.line ?? 0,
originalSnippet: chosen.name,
mutatedSnippet: mutated,
identifier: { correct: chosen.name, misspelled: mutated },
},
];
}
@@ -928,6 +899,7 @@ class IdentifierMultiEditMutation extends BaseAstMutation {
lineNumber: selectedPaths[0]?.node.loc?.start.line ?? 0,
originalSnippet: chosen.name,
mutatedSnippet: mutated,
identifier: { correct: chosen.name, misspelled: mutated },
};
}
}
@@ -935,8 +907,6 @@ class IdentifierMultiEditMutation extends BaseAstMutation {
class DuplicateLineLiteralFlipMutation extends BaseAstMutation {
name = "duplicate-line-flip";
category = "duplicate";
fixHint = "Fix the literal or operator on the duplicated line.";
description = "A duplicated line contains a subtle literal/operator change.";
collectCandidates(parsed: Parsed): Candidate<t.Statement, { group: string }>[] {
const out: Candidate<t.Statement, { group: string }>[] = [];
@@ -1025,8 +995,6 @@ class DuplicateLineLiteralFlipMutation extends BaseAstMutation {
class SwapAdjacentLinesMutation extends BaseAstMutation {
name = "swap-adjacent-lines";
category = "structural";
fixHint = "Swap the two adjacent lines back to their original order.";
description = "Two adjacent statements are in the wrong order.";
collectCandidates(parsed: Parsed): Candidate<t.Program | t.BlockStatement, { index: number }>[] {
const out: Candidate<t.Program | t.BlockStatement, { index: number }>[] = [];
@@ -1134,8 +1102,6 @@ class SwapAdjacentLinesMutation extends BaseAstMutation {
class SwapIfElseBranchesMutation extends BaseAstMutation {
name = "swap-if-else";
category = "structural";
fixHint = "Swap the if and else branch bodies back to their original positions.";
description = "The if and else branches are swapped.";
collectCandidates(parsed: Parsed): Candidate<t.IfStatement>[] {
const out: Candidate<t.IfStatement>[] = [];
@@ -1163,29 +1129,31 @@ class SwapIfElseBranchesMutation extends BaseAstMutation {
}
}
class RemoveEarlyReturnMutation extends BaseAstMutation {
name = "remove-early-return";
/**
* Remove one label from a fall-through `case A: case B:` pair. The fix inserts
* the missing label back — a deterministic pure-insertion task whose content
* is dictated by the visible sibling label, not by hidden deleted code.
*/
class RemoveCaseLabelMutation extends BaseAstMutation {
name = "remove-case-label";
category = "structural";
fixHint =
"Restore the missing guard clause (if statement with early return). Add back the exact 3-line pattern: if condition, return statement, closing brace.";
description = "A guard clause (early return) was removed.";
collectCandidates(parsed: Parsed): Candidate<t.IfStatement>[] {
const out: Candidate<t.IfStatement>[] = [];
collectCandidates(parsed: Parsed): Candidate<t.SwitchCase>[] {
const out: Candidate<t.SwitchCase>[] = [];
traverse(parsed.ast, {
IfStatement: path => {
SwitchCase: path => {
const node = path.node;
if (node.alternate) return;
if (!t.isBlockStatement(node.consequent)) return;
if (node.consequent.body.length !== 1) return;
if (!t.isReturnStatement(node.consequent.body[0])) return;
if (!node.test || node.consequent.length > 0) return;
if (!t.isSwitchStatement(path.parent)) return;
const index = path.parent.cases.indexOf(node);
if (index < 0 || index >= path.parent.cases.length - 1) return;
out.push({ path });
},
});
return out;
}
applyCandidate(parsed: Parsed, candidate: Candidate<t.IfStatement>): MutationInfo {
applyCandidate(parsed: Parsed, candidate: Candidate<t.SwitchCase>): MutationInfo {
const node = candidate.path.node;
const before = snippetFromSource(parsed.code, node, snippetFromNode(node));
candidate.path.remove();
@@ -1193,131 +1161,331 @@ class RemoveEarlyReturnMutation extends BaseAstMutation {
}
}
class SwapNamedImportsMutation extends BaseAstMutation {
name = "swap-named-imports";
category = "import";
fixHint =
"Swap ONLY the two imported names that are in the wrong order. Do not reorder other imports or modify other import statements.";
description = "Two named imports are swapped in a destructuring import.";
/**
* Wrap a block's body in a redundant `if (true) { ... }`. The fix removes the
* wrapper and dedents — an indentation-shifting multi-line edit whose entire
* content stays visible in the buggy file (no invisible-code reconstruction).
*/
class WrapRedundantIfMutation extends BaseAstMutation {
name = "wrap-redundant-if";
category = "structural";
collectCandidates(parsed: Parsed): Candidate<t.ImportDeclaration, { i: number; j: number }>[] {
const out: Candidate<t.ImportDeclaration, { i: number; j: number }>[] = [];
#lineSpan(node: t.Node): number {
if (!node.loc) return 0;
return node.loc.end.line - node.loc.start.line + 1;
}
collectCandidates(parsed: Parsed): Candidate<t.BlockStatement>[] {
const out: Candidate<t.BlockStatement>[] = [];
traverse(parsed.ast, {
ImportDeclaration: path => {
const named = path.node.specifiers
.map((spec, idx) => ({ spec, idx }))
.filter((entry): entry is { spec: t.ImportSpecifier; idx: number } => t.isImportSpecifier(entry.spec))
.filter(({ spec }) => t.isIdentifier(spec.imported) && t.isIdentifier(spec.local))
.filter(
({ spec }) =>
t.isIdentifier(spec.imported) &&
t.isIdentifier(spec.local) &&
spec.imported.name === spec.local.name,
);
if (named.length < 2) return;
for (let i = 0; i < named.length; i++) {
for (let j = i + 1; j < named.length; j++) {
out.push({ path, meta: { i: named[i]!.idx, j: named[j]!.idx } });
}
}
BlockStatement: path => {
const span = this.#lineSpan(path.node);
if (span < 4 || span > 160) return;
if (path.node.body.length === 0) return;
if (path.node.directives.length > 0) return;
// Don't wrap a body that is itself just an `if (true)` (repeated mutation noise).
if (path.node.body.length === 1 && t.isIfStatement(path.node.body[0])) return;
out.push({ path });
},
});
return out;
}
applyCandidate(_parsed: Parsed, candidate: Candidate<t.BlockStatement>): MutationInfo {
const node = candidate.path.node;
const line = nodeLine(node);
node.body = [t.ifStatement(t.booleanLiteral(true), t.blockStatement(node.body))];
return { lineNumber: line, originalSnippet: "", mutatedSnippet: "if (true) {" };
}
}
/**
* Swap two adjacent multi-line sibling statements (functions, if-chains,
* loops). The fix swaps them back — a large contiguous replace whose content
* is fully visible.
*/
class SwapSiblingBlocksMutation implements Mutation {
name = "swap-sibling-blocks";
category = "structural";
#collect(parsed: Parsed): Array<{ left: t.Statement; right: t.Statement }> {
const out: Array<{ left: t.Statement; right: t.Statement }> = [];
const lineSpan = (node: t.Node): number => (node.loc ? node.loc.end.line - node.loc.start.line + 1 : 0);
const considerList = (body: Array<t.Statement | t.ModuleDeclaration>): void => {
for (let i = 0; i < body.length - 1; i++) {
const left = body[i];
const right = body[i + 1];
if (!left || !right || !t.isStatement(left) || !t.isStatement(right)) continue;
if (!left.loc || !right.loc) continue;
if (t.isImportDeclaration(left) || t.isImportDeclaration(right)) continue;
const leftSpan = lineSpan(left);
const rightSpan = lineSpan(right);
// At least one true block; bounded total so prompts stay readable.
if (Math.max(leftSpan, rightSpan) < 3) continue;
if (leftSpan > 80 || rightSpan > 80 || leftSpan + rightSpan > 120) continue;
const gap = right.loc.start.line - left.loc.end.line;
if (gap > 2) continue;
const leftText = snippetFromSource(parsed.code, left, "").trim();
const rightText = snippetFromSource(parsed.code, right, "").trim();
if (!leftText || !rightText || leftText === rightText) continue;
out.push({ left, right });
}
};
traverse(parsed.ast, {
Program: path => considerList(path.node.body),
BlockStatement: path => considerList(path.node.body),
});
return out;
}
canApply(content: string): boolean {
const parsed = parseCode(content);
return parsed !== null && this.#collect(parsed).length > 0;
}
mutate(content: string, rng: () => number): [string, MutationInfo] {
const parsed = parseCode(content);
if (!parsed) return [content, noopInfo()];
const candidates = this.collectCandidates(parsed);
const candidates = this.#collect(parsed);
if (candidates.length === 0) return [content, noopInfo()];
const chosen = randomChoice(candidates, rng);
const node = chosen.path.node;
const indices = chosen.meta;
if (!indices) return [content, noopInfo()];
const { i, j } = indices;
if (i < 0 || j < 0 || i >= node.specifiers.length || j >= node.specifiers.length) return [content, noopInfo()];
const left = node.specifiers[i];
const right = node.specifiers[j];
if (!left || !right) return [content, noopInfo()];
const { left, right } = randomChoice(candidates, rng);
const leftRange = nodeRange(left);
const rightRange = nodeRange(right);
const importRange = nodeRange(node);
if (!leftRange || !rightRange || !importRange) return [content, noopInfo()];
if (leftRange.end > rightRange.start) return [content, noopInfo()];
if (!leftRange || !rightRange || leftRange.end > rightRange.start) return [content, noopInfo()];
const leftText = content.slice(leftRange.start, leftRange.end);
const rightText = content.slice(rightRange.start, rightRange.end);
const between = content.slice(leftRange.end, rightRange.start);
const swapped = `${content.slice(rightRange.start, rightRange.end)}${between}${content.slice(leftRange.start, leftRange.end)}`;
const mutated = applySourceEdits(content, [
{ start: leftRange.start, end: leftRange.end, replacement: rightText },
{ start: rightRange.start, end: rightRange.end, replacement: leftText },
{ start: leftRange.start, end: rightRange.end, replacement: swapped },
]);
if (!mutated || mutated === content) return [content, noopInfo()];
return [
mutated,
{
lineNumber: nodeLine(node),
originalSnippet: content.slice(importRange.start, importRange.end).trim(),
mutatedSnippet: mutated.slice(importRange.start, importRange.end).trim(),
lineNumber: left.loc?.start.line ?? 0,
originalSnippet: `lines ${left.loc?.start.line ?? 0}-${right.loc?.end.line ?? 0}`,
mutatedSnippet: "[swapped]",
},
];
}
applyCandidate(parsed: Parsed, candidate: Candidate<t.ImportDeclaration, { i: number; j: number }>): MutationInfo {
const node = candidate.path.node;
const indices = candidate.meta;
if (!indices) return noopInfo();
const before = snippetFromSource(parsed.code, node, snippetFromNode(node));
const { i, j } = indices;
if (i < 0 || j < 0 || i >= node.specifiers.length || j >= node.specifiers.length) return noopInfo();
[node.specifiers[i], node.specifiers[j]] = [node.specifiers[j]!, node.specifiers[i]!];
return {
lineNumber: nodeLine(node),
originalSnippet: before.trim(),
mutatedSnippet: snippetFromNode(node).trim(),
};
}
}
class DeleteStatementMutation extends BaseAstMutation {
name = "delete-statement";
/**
* Duplicate a multi-line statement right after itself (a copy-paste accident).
* The fix deletes the second copy — a large deletion where the surviving copy
* stays visible, so the task is deterministic without revealing hidden code.
*/
class DuplicateBlockMutation implements Mutation {
name = "duplicate-block";
category = "structural";
fixHint = "Restore the deleted statement.";
description = "A critical statement was deleted from the code.";
collectCandidates(parsed: Parsed): Candidate<t.Statement>[] {
const out: Candidate<t.Statement>[] = [];
#collect(parsed: Parsed): t.Statement[] {
const out: t.Statement[] = [];
const consider = (statement: t.Statement | t.ModuleDeclaration): void => {
if (!t.isStatement(statement) || !statement.loc) return;
const span = statement.loc.end.line - statement.loc.start.line + 1;
if (span < 3 || span > 120) return;
// Duplicating lexical declarations or exports produces parse/redeclaration
// errors; stick to statements that stay syntactically valid twice.
if (
!t.isIfStatement(statement) &&
!t.isExpressionStatement(statement) &&
!t.isForStatement(statement) &&
!t.isForOfStatement(statement) &&
!t.isWhileStatement(statement) &&
!t.isTryStatement(statement) &&
!t.isSwitchStatement(statement)
) {
return;
}
out.push(statement);
};
traverse(parsed.ast, {
Statement: path => {
if (!path.node.loc) return;
if (t.isVariableDeclaration(path.node)) {
out.push({ path: path as NodePath<t.Statement> });
return;
}
if (!t.isExpressionStatement(path.node)) return;
if (t.isAssignmentExpression(path.node.expression) || t.isUpdateExpression(path.node.expression)) {
out.push({ path: path as NodePath<t.Statement> });
}
Program: path => {
for (const statement of path.node.body) consider(statement);
},
BlockStatement: path => {
for (const statement of path.node.body) consider(statement);
},
});
return out;
}
applyCandidate(parsed: Parsed, candidate: Candidate<t.Statement>): MutationInfo {
const node = candidate.path.node;
const before = snippetFromSource(parsed.code, node, snippetFromNode(node));
candidate.path.remove();
return { lineNumber: nodeLine(node), originalSnippet: before.trim(), mutatedSnippet: "[removed]" };
canApply(content: string): boolean {
const parsed = parseCode(content);
return parsed !== null && this.#collect(parsed).length > 0;
}
mutate(content: string, rng: () => number): [string, MutationInfo] {
const parsed = parseCode(content);
if (!parsed) return [content, noopInfo()];
const candidates = this.#collect(parsed);
if (candidates.length === 0) return [content, noopInfo()];
const statement = randomChoice(candidates, rng);
const range = nodeRange(statement);
if (!range) return [content, noopInfo()];
const text = content.slice(range.start, range.end);
const mutated = applySourceEdits(content, [{ start: range.end, end: range.end, replacement: `\n\n${text}` }]);
if (!mutated || mutated === content) return [content, noopInfo()];
// Reject mutations that no longer parse (e.g. duplicated declarations).
if (!parseCode(mutated)) return [content, noopInfo()];
return [
mutated,
{
lineNumber: statement.loc?.start.line ?? 0,
originalSnippet: text.split("\n")[0]?.trim() ?? "",
mutatedSnippet: text.split("\n")[0]?.trim() ?? "",
},
];
}
}
/**
* Move a multi-line statement to a distant position in the same statement
* list. The fix moves it back — one delete hunk plus one insert hunk, the two
* dominant hunk shapes in real edits — with the moved content fully visible.
*/
class MoveDistantBlockMutation implements Mutation {
name = "move-distant-block";
category = "structural";
multiHunk = true;
#collect(parsed: Parsed): Array<{ moved: t.Statement; next: t.Statement; target: t.Statement }> {
const out: Array<{ moved: t.Statement; next: t.Statement; target: t.Statement }> = [];
const considerList = (body: Array<t.Statement | t.ModuleDeclaration>): void => {
if (body.length < 5) return;
for (let from = 0; from < body.length - 1; from++) {
const moved = body[from];
const next = body[from + 1];
if (!moved || !next || !t.isStatement(moved) || !t.isStatement(next)) continue;
if (!moved.loc || t.isImportDeclaration(moved)) continue;
const span = moved.loc.end.line - moved.loc.start.line + 1;
if (span < 3 || span > 60) continue;
for (let to = from + 3; to < body.length; to++) {
const target = body[to];
if (!target || !t.isStatement(target) || !target.loc) continue;
out.push({ moved, next, target });
}
}
};
traverse(parsed.ast, {
Program: path => considerList(path.node.body),
BlockStatement: path => considerList(path.node.body),
});
return out;
}
canApply(content: string): boolean {
const parsed = parseCode(content);
return parsed !== null && this.#collect(parsed).length > 0;
}
mutate(content: string, rng: () => number): [string, MutationInfo] {
const parsed = parseCode(content);
if (!parsed) return [content, noopInfo()];
const candidates = this.#collect(parsed);
if (candidates.length === 0) return [content, noopInfo()];
const { moved, next, target } = randomChoice(candidates, rng);
const movedRange = nodeRange(moved);
const nextRange = nodeRange(next);
const targetRange = nodeRange(target);
if (!movedRange || !nextRange || !targetRange) return [content, noopInfo()];
if (movedRange.end > nextRange.start || nextRange.start > targetRange.end) return [content, noopInfo()];
const movedText = content.slice(movedRange.start, movedRange.end);
const mutated = applySourceEdits(content, [
// Cut the statement together with its trailing separator...
{ start: movedRange.start, end: nextRange.start, replacement: "" },
// ...and splice it back in after the distant target statement.
{ start: targetRange.end, end: targetRange.end, replacement: `\n\n${movedText}` },
]);
if (!mutated || mutated === content) return [content, noopInfo()];
if (!parseCode(mutated)) return [content, noopInfo()];
return [
mutated,
{
lineNumber: moved.loc?.start.line ?? 0,
originalSnippet: movedText.split("\n")[0]?.trim() ?? "",
mutatedSnippet: movedText.split("\n")[0]?.trim() ?? "",
},
];
}
}
/**
* Apply several independent token-level mutations in one file — the multi-hunk
* edit shape that dominates real sessions. Each constituent bug is a
* single-line change fully specified by the task's before/after blocks.
*/
class CompositeMultiEditMutation implements Mutation {
name = "composite-multi-edit";
category = "multi";
multiHunk = true;
#parts: Mutation[];
constructor(parts: Mutation[]) {
this.#parts = parts;
}
canApply(content: string): boolean {
let applicable = 0;
for (const part of this.#parts) {
try {
if (part.canApply(content)) applicable++;
} catch {
// Unparseable for this part; skip.
}
if (applicable >= 2) return true;
}
return false;
}
mutate(content: string, rng: () => number): [string, MutationInfo] {
const applicable = this.#parts.filter(part => {
try {
return part.canApply(content);
} catch {
return false;
}
});
if (applicable.length < 2) return [content, noopInfo()];
const target = Math.min(3 + Math.floor(rng() * 3), applicable.length);
const parts = randomSample(applicable, target, rng);
let current = content;
let firstInfo: MutationInfo | null = null;
let applied = 0;
for (const part of parts) {
try {
const [next, info] = part.mutate(current, rng);
if (next === current || info.lineNumber === 0) continue;
current = next;
firstInfo ??= info;
applied++;
} catch {
// A part failing on already-mutated content just shrinks the composite.
}
}
if (applied < 2 || !firstInfo) return [content, noopInfo()];
return [current, firstInfo];
}
}
class OffByOneMutation extends BaseAstMutation {
name = "off-by-one";
category = "literal";
fixHint = "Fix the off-by-one error in the numeric literal or comparison.";
description = "A numeric boundary has an off-by-one error.";
collectCandidates(parsed: Parsed): Candidate<t.NumericLiteral | t.BinaryExpression>[] {
const out: Candidate<t.NumericLiteral | t.BinaryExpression>[] = [];
@@ -1386,7 +1554,8 @@ class OffByOneMutation extends BaseAstMutation {
}
}
export const ALL_MUTATIONS: Mutation[] = [
/** Single-line token mutations; also the constituent parts of the composite. */
const TOKEN_MUTATIONS: Mutation[] = [
new SwapComparisonMutation(),
new SwapEqualityMutation(),
new SwapLogicalMutation(),
@@ -1399,14 +1568,21 @@ export const ALL_MUTATIONS: Mutation[] = [
new NullishCoalescingSwapMutation(),
new RegexQuantifierSwapMutation(),
new UnicodeHyphenMutation(),
new OffByOneMutation(),
];
export const ALL_MUTATIONS: Mutation[] = [
...TOKEN_MUTATIONS,
new IdentifierMultiEditMutation(),
new DuplicateLineLiteralFlipMutation(),
new SwapAdjacentLinesMutation(),
new SwapIfElseBranchesMutation(),
new RemoveEarlyReturnMutation(),
new SwapNamedImportsMutation(),
new DeleteStatementMutation(),
new OffByOneMutation(),
new WrapRedundantIfMutation(),
new SwapSiblingBlocksMutation(),
new DuplicateBlockMutation(),
new MoveDistantBlockMutation(),
new RemoveCaseLabelMutation(),
new CompositeMultiEditMutation(TOKEN_MUTATIONS),
];
export const CATEGORY_MAP: Record<string, string[]> = {
@@ -1419,5 +1595,5 @@ export const CATEGORY_MAP: Record<string, string[]> = {
identifier: ALL_MUTATIONS.filter(m => m.category === "identifier").map(m => m.name),
duplicate: ALL_MUTATIONS.filter(m => m.category === "duplicate").map(m => m.name),
structural: ALL_MUTATIONS.filter(m => m.category === "structural").map(m => m.name),
import: ALL_MUTATIONS.filter(m => m.category === "import").map(m => m.name),
multi: ALL_MUTATIONS.filter(m => m.category === "multi").map(m => m.name),
};
@@ -0,0 +1,9 @@
# Fix a misspelled identifier in `{{filename}}`
A recent edit misspelled the identifier `{{correct}}` as `{{misspelled}}` in {{#when count "==" 1}}one place{{else}}{{count}} places{{/when}}.
{{#if affectedLines}}
Affected line{{#when count ">" 1}}s{{/when}}: {{join affectedLines ", "}}.
{{/if}}
Replace every occurrence of `{{misspelled}}` with `{{correct}}`. Do not change anything else.
@@ -0,0 +1,85 @@
# Fix a bug in `{{filename}}`
{{#when name "==" "swap-comparison"}}
A comparison operator is subtly wrong.
{{/when}}
{{#when name "==" "swap-equality"}}
An equality operator is inverted.
{{/when}}
{{#when name "==" "swap-logical"}}
A boolean operator is incorrect.
{{/when}}
{{#when name "==" "remove-negation"}}
A logical negation (`!`) was accidentally removed.
{{/when}}
{{#when name "==" "swap-increment-decrement"}}
An increment/decrement operator points the wrong direction.
{{/when}}
{{#when name "==" "swap-arithmetic"}}
An arithmetic operator was swapped.
{{/when}}
{{#when name "==" "flip-boolean"}}
A boolean literal is inverted.
{{/when}}
{{#when name "==" "remove-optional-chain"}}
Optional chaining was removed from a property access.
{{/when}}
{{#when name "==" "swap-call-args"}}
Two arguments in a call are swapped.
{{/when}}
{{#when name "==" "swap-nullish"}}
A nullish coalescing operator was swapped.
{{/when}}
{{#when name "==" "swap-regex-quantifier"}}
A regex quantifier was swapped, changing whitespace matching.
{{/when}}
{{#when name "==" "unicode-hyphen"}}
A string literal contains a lookalike unicode dash.
{{/when}}
{{#when name "==" "off-by-one"}}
A numeric boundary has an off-by-one error.
{{/when}}
{{#when name "==" "duplicate-line-flip"}}
One copy of a line that repeats elsewhere in this file was altered.
{{/when}}
{{#when name "==" "composite-multi-edit"}}
This file contains several small, unrelated single-line bugs.
{{/when}}
{{#if functionName}}
The bug is in the `{{functionName}}` function.
{{/if}}
{{#when region "==" "top"}}
The bug is near the top of the file.
{{/when}}
{{#when region "==" "middle"}}
The bug is around the middle of the file.
{{/when}}
{{#when region "==" "end"}}
The bug is near the end of the file.
{{/when}}
{{#each hunks}}
{{#when ../hunkCount ">" 1}}
## Change {{add @index 1}}
{{/when}}
{{#if isDelete}}Delete this block{{#if startLine}} (starting on line {{startLine}}){{/if}}:{{else}}Replace this{{#if startLine}} (starting on line {{startLine}}){{/if}}:{{/if}}
{{../fence}}{{../language}}
{{oldCode}}
{{../fence}}
{{#unless isDelete}}
with:
{{../fence}}{{../language}}
{{newCode}}
{{../fence}}
{{/unless}}
{{/each}}
{{#if nightmare}}This file contains near-identical code in multiple places — edit exactly the block shown and nothing else.{{else}}Make exactly this change; do not modify anything else.{{/if}}
@@ -0,0 +1,37 @@
# Fix a bug in `{{filename}}`
{{#when kind "==" "case-label"}}
In this file's `switch`, the value in `{{label}}` must be handled exactly like the case after it: add a fall-through `{{label}}` label directly before `{{before}}`.
{{/when}}
{{#when kind "==" "duplicate-block"}}
The block starting with `{{head}}` appears twice in a row — the second copy is a copy-paste accident. Delete the second copy and keep the first.
{{/when}}
{{#when kind "==" "move-block"}}
The block starting with `{{head}}` was moved to the wrong place — it currently sits after `{{currentPrev}}`. Move it back so it comes directly before `{{destination}}`.
{{/when}}
{{#when kind "==" "wrap-if"}}
A leftover debugging wrapper is redundant: remove the `if (true) {` on line {{wrapperLine}} together with its closing brace, and dedent the wrapped body one level.
{{/when}}
{{#when kind "==" "swap-blocks"}}
Two adjacent blocks are in the wrong order: the block starting with `{{secondHead}}` belongs before the block starting with `{{firstHead}}`. Swap the two blocks.
{{/when}}
{{#when kind "==" "swap-lines"}}
Two adjacent statements are in the wrong order: `{{secondHead}}` belongs before `{{firstHead}}`. Swap the two statements.
{{/when}}
{{#when kind "==" "swap-if-else"}}
The branch bodies of `{{condition}}` are swapped: the current `else` body belongs under the `if`, and vice versa. Swap the two branch bodies.
{{/when}}
After the fix, the affected {{#when hunkCount ">" 1}}regions must{{else}}region must{{/when}} read exactly:
{{#each hunks}}
{{#if startLine}}
Around line {{startLine}}:
{{/if}}
{{../fence}}{{../language}}
{{newCode}}
{{../fence}}
{{/each}}
Make exactly this change; do not modify anything else.
@@ -0,0 +1,96 @@
import { describe, expect, it } from "bun:test";
import {
placementsFromDiff,
type RenderedHunk,
renderHunks,
solveRenderedHunks,
} from "@oh-my-pi/typescript-edit-benchmark/hunks";
/** Round-trip an input/expected pair through the prompt machinery. */
function roundTrip(input: string, expected: string): { hunks: RenderedHunk[]; solved: string } {
const inputLines = input.replace(/\n$/, "").split("\n");
const placements = placementsFromDiff(input, expected);
const hunks = renderHunks(inputLines, placements);
if (!hunks) throw new Error("renderHunks returned null");
const solved = solveRenderedHunks(inputLines, hunks);
if (!solved) throw new Error("solveRenderedHunks returned null");
return { hunks, solved: `${solved.join("\n")}\n` };
}
describe("hunk round-trip", () => {
it("reproduces the expected file for a block replacement", () => {
const input = "const a = 1;\nfunction f() {\n\treturn a;\n}\nexport { f };\n";
const expected = "const a = 1;\nfunction f() {\n\treturn a + 1;\n}\nexport { f };\n";
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
expect(hunks).toHaveLength(1);
expect(hunks[0].unique).toBe(true);
});
it("reproduces the expected file for a pure insertion, anchored on a non-blank line", () => {
const input = "function f() {\n\n\treturn 1;\n}\n";
const expected = "function f() {\n\n\tconst x = 0;\n\treturn 1;\n}\n";
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
// The line directly above the insertion is blank; the anchor must be visible.
expect(hunks[0].oldBlock.some(line => line.trim().length > 0)).toBe(true);
});
it("reproduces the expected file for a pure deletion, anchored on context", () => {
const input = "a();\nb();\nc();\n";
const expected = "a();\nc();\n";
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
// Deletions keep a context anchor so the replacement is never an empty fence.
expect(hunks[0].newBlock.length).toBeGreaterThan(0);
expect(hunks[0].oldBlock.length).toBeGreaterThan(hunks[0].newBlock.length);
});
});
describe("uniqueness handling", () => {
it("extends a repeated block with context until it is unique", () => {
const input = "start();\nif (x) {\n\tstop();\n}\nif (y) {\n\tstop();\n}\n";
const expected = "start();\nif (x) {\n\tstop();\n}\nif (y) {\n\thalt();\n}\n";
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
expect(hunks[0].unique).toBe(true);
// `\tstop();` alone is ambiguous; the block must carry the `if (y) {` context.
expect(hunks[0].oldBlock.length).toBeGreaterThan(1);
});
it("marks a hunk non-unique with its start line when context cannot disambiguate", () => {
const line = "value += 1;";
const input = `${Array.from({ length: 4 }, () => line).join("\n")}\n`;
const expected = `${[line, line, "value += 2;", line].join("\n")}\n`;
const inputLines = input.replace(/\n$/, "").split("\n");
const hunks = renderHunks(inputLines, placementsFromDiff(input, expected));
if (!hunks) throw new Error("renderHunks returned null");
expect(hunks[0].unique).toBe(false);
// startLine must locate a real occurrence of the (extended) old block, so
// a solver that trusts it lands on the right copy.
expect(inputLines.slice(hunks[0].startLine - 1, hunks[0].startLine - 1 + hunks[0].oldBlock.length)).toEqual(
hunks[0].oldBlock,
);
const solved = solveRenderedHunks(inputLines, hunks);
expect(solved ? `${solved.join("\n")}\n` : null).toBe(expected);
});
});
describe("placement merging", () => {
it("renders two adjacent swapped statements as a single hunk", () => {
const input = "setup();\nsecond();\nfirst();\nteardown();\n";
const expected = "setup();\nfirst();\nsecond();\nteardown();\n";
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
expect(hunks).toHaveLength(1);
});
it("keeps distant changes as separate hunks", () => {
const filler = Array.from({ length: 10 }, (_, i) => `line${i}();`).join("\n");
const input = `alpha();\n${filler}\nomega();\n`;
const expected = `alpha2();\n${filler}\nomega2();\n`;
const { hunks, solved } = roundTrip(input, expected);
expect(solved).toBe(expected);
expect(hunks).toHaveLength(2);
});
});