chore: expanded Biome configuration to include TypeScript files in package root directories
- Expanded Biome configuration to include TypeScript files in package root directories with new glob pattern `packages/*/*.ts`.
This commit is contained in:
@@ -39,6 +39,7 @@
|
||||
"packages/*/src/**/*.tsx",
|
||||
"packages/*/test/**/*.ts",
|
||||
"packages/*/examples/**/*.ts",
|
||||
"packages/*/*.ts",
|
||||
"!**/vendor/**/*",
|
||||
"!**/node_modules/**/*",
|
||||
"!**/test-sessions.ts",
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added whitespace normalization in line reference parsing to tolerate spaces around colons (e.g., `5 : ab` now parses as `5:ab`)
|
||||
@@ -12,6 +11,10 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Reverted hash algorithm from 3-character base-36 back to 2-character hexadecimal for line references
|
||||
- Enhanced range validation during hashline edits to detect and reject relocations that change the scope of affected lines
|
||||
- Improved wrapped-line restoration logic to only attempt merging when source lines exhibit continuation patterns
|
||||
- Updated hashline tool documentation to emphasize direction-locking mutations and clarify recovery procedures for hash mismatches
|
||||
- Changed `applyHashlineEdits` return type to include optional `warnings` array for reporting suspicious edit patterns
|
||||
- Improved hash relocation logic to recompute touched lines after hash-based line number adjustments, preventing incorrect merge heuristics
|
||||
- Enhanced error messages for no-op edits to include preview of target lines with their current hashes and content
|
||||
@@ -31,6 +34,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed range-based edits to prevent invalid mutations when hash relocation changes the number of lines in the target range
|
||||
- Fixed multi-edit application to use original file state for all anchor references, preventing incorrect line numbers when earlier edits change file length
|
||||
|
||||
## [11.10.4] - 2026-02-10
|
||||
|
||||
@@ -280,8 +280,8 @@ function stripNewLinePrefixes(lines: string[]): string[] {
|
||||
});
|
||||
}
|
||||
|
||||
const HASH_LEN = 3;
|
||||
const RADIX = 36;
|
||||
const HASH_LEN = 2;
|
||||
const RADIX = 16;
|
||||
const HASH_MOD = RADIX ** HASH_LEN;
|
||||
|
||||
const DICT = Array.from({ length: HASH_MOD }, (_, i) => i.toString(RADIX).padStart(HASH_LEN, "0"));
|
||||
@@ -741,41 +741,76 @@ export function applyHashlineEdits(
|
||||
}
|
||||
uniqueLineByHash.set(hash, lineNo);
|
||||
}
|
||||
|
||||
function buildMismatch(ref: { line: number; hash: string }, line = ref.line): HashMismatch {
|
||||
return {
|
||||
line,
|
||||
expected: ref.hash,
|
||||
actual: computeLineHash(line, fileLines[line - 1]),
|
||||
};
|
||||
}
|
||||
|
||||
function validateOrRelocateRef(ref: {
|
||||
line: number;
|
||||
hash: string;
|
||||
}): { ok: true; relocated: boolean } | { ok: false } {
|
||||
if (ref.line < 1 || ref.line > fileLines.length) {
|
||||
throw new Error(`Line ${ref.line} does not exist (file has ${fileLines.length} lines)`);
|
||||
}
|
||||
const expected = ref.hash.toLowerCase();
|
||||
const actualHash = computeLineHash(ref.line, fileLines[ref.line - 1]);
|
||||
if (actualHash === expected) {
|
||||
return { ok: true, relocated: false };
|
||||
}
|
||||
|
||||
const relocated = uniqueLineByHash.get(expected);
|
||||
if (relocated === undefined) {
|
||||
mismatches.push({ line: ref.line, expected: ref.hash, actual: actualHash });
|
||||
return { ok: false };
|
||||
}
|
||||
ref.line = relocated;
|
||||
return { ok: true, relocated: true };
|
||||
}
|
||||
for (const { spec, dstLines } of parsed) {
|
||||
const refsToValidate: { line: number; hash: string }[] = [];
|
||||
switch (spec.kind) {
|
||||
case "single":
|
||||
refsToValidate.push(spec.ref);
|
||||
case "single": {
|
||||
const status = validateOrRelocateRef(spec.ref);
|
||||
if (!status.ok) continue;
|
||||
break;
|
||||
case "range":
|
||||
if (spec.start.line > spec.end.line) {
|
||||
throw new Error(`Range start line ${spec.start.line} must be <= end line ${spec.end.line}`);
|
||||
}
|
||||
refsToValidate.push(spec.start, spec.end);
|
||||
break;
|
||||
case "insertAfter":
|
||||
}
|
||||
case "insertAfter": {
|
||||
if (dstLines.length === 0) {
|
||||
throw new Error('Insert-after edit (src "N:HH..") requires non-empty dst');
|
||||
}
|
||||
refsToValidate.push(spec.after);
|
||||
const status = validateOrRelocateRef(spec.after);
|
||||
if (!status.ok) continue;
|
||||
break;
|
||||
}
|
||||
}
|
||||
case "range": {
|
||||
if (spec.start.line > spec.end.line) {
|
||||
throw new Error(`Range start line ${spec.start.line} must be <= end line ${spec.end.line}`);
|
||||
}
|
||||
|
||||
for (const ref of refsToValidate) {
|
||||
if (ref.line < 1 || ref.line > fileLines.length) {
|
||||
throw new Error(`Line ${ref.line} does not exist (file has ${fileLines.length} lines)`);
|
||||
}
|
||||
const actualHash = computeLineHash(ref.line, fileLines[ref.line - 1]);
|
||||
if (actualHash === ref.hash.toLowerCase()) {
|
||||
continue;
|
||||
}
|
||||
const originalStart = spec.start.line;
|
||||
const originalEnd = spec.end.line;
|
||||
const originalCount = originalEnd - originalStart + 1;
|
||||
|
||||
const relocated = uniqueLineByHash.get(ref.hash.toLowerCase());
|
||||
if (relocated !== undefined) {
|
||||
ref.line = relocated;
|
||||
continue;
|
||||
const startStatus = validateOrRelocateRef(spec.start);
|
||||
const endStatus = validateOrRelocateRef(spec.end);
|
||||
if (!startStatus.ok || !endStatus.ok) continue;
|
||||
|
||||
const relocatedCount = spec.end.line - spec.start.line + 1;
|
||||
const changedByRelocation = startStatus.relocated || endStatus.relocated;
|
||||
const invalidRange = spec.start.line > spec.end.line;
|
||||
const scopeChanged = relocatedCount !== originalCount;
|
||||
|
||||
if (changedByRelocation && (invalidRange || scopeChanged)) {
|
||||
spec.start.line = originalStart;
|
||||
spec.end.line = originalEnd;
|
||||
mismatches.push(buildMismatch(spec.start, originalStart), buildMismatch(spec.end, originalEnd));
|
||||
}
|
||||
break;
|
||||
}
|
||||
mismatches.push({ line: ref.line, expected: ref.hash, actual: actualHash });
|
||||
}
|
||||
}
|
||||
|
||||
@@ -916,13 +951,12 @@ export function applyHashlineEdits(
|
||||
const origCanon = stripAllWhitespace(orig);
|
||||
const origCanonForMatch = stripTrailingContinuationTokens(origCanon);
|
||||
const origCanonForMergeOps = stripMergeOperatorChars(origCanon);
|
||||
const origLooksLikeContinuation = origCanonForMatch.length < origCanon.length;
|
||||
if (origCanon.length === 0) return null;
|
||||
|
||||
const prevIdx = line - 2;
|
||||
const nextIdx = line;
|
||||
|
||||
const prevIdx = line - 2;
|
||||
// Case A: dst absorbed the next continuation line.
|
||||
if (nextIdx < fileLines.length && !explicitlyTouchedLines.has(line + 1)) {
|
||||
if (origLooksLikeContinuation && nextIdx < fileLines.length && !explicitlyTouchedLines.has(line + 1)) {
|
||||
const next = fileLines[nextIdx];
|
||||
const nextCanon = stripAllWhitespace(next);
|
||||
const a = newCanon.indexOf(origCanonForMatch);
|
||||
@@ -931,12 +965,13 @@ export function applyHashlineEdits(
|
||||
return { startLine: line, deleteCount: 2, newLines: [newLine] };
|
||||
}
|
||||
}
|
||||
|
||||
// Case B: dst absorbed the previous declaration/continuation line.
|
||||
if (prevIdx >= 0 && !explicitlyTouchedLines.has(line - 1)) {
|
||||
const prev = fileLines[prevIdx];
|
||||
const prevCanon = stripAllWhitespace(prev);
|
||||
const prevCanonForMatch = stripTrailingContinuationTokens(prevCanon);
|
||||
const prevLooksLikeContinuation = prevCanonForMatch.length < prevCanon.length;
|
||||
if (!prevLooksLikeContinuation) return null;
|
||||
const a = newCanonForMergeOps.indexOf(stripMergeOperatorChars(prevCanonForMatch));
|
||||
const b = newCanonForMergeOps.indexOf(origCanonForMergeOps);
|
||||
if (a !== -1 && b !== -1 && a < b && newCanon.length <= prevCanon.length + origCanon.length + 32) {
|
||||
|
||||
@@ -9,7 +9,7 @@ Line-addressed edits using hash-verified line references. Read file with hashes
|
||||
- If you already edited a file in this turn, re-read that file before the next edit to it
|
||||
- For code-change requests, respond with tool calls, not prose
|
||||
- Edit only requested lines. Do not reformat unrelated code.
|
||||
- Do not submit a replacement whose content is identical to the current line. If unsure, re-read the target lines first.
|
||||
- Direction-lock every mutation: replace the exact currently-present token/expression with the intended target token/expression; never reverse the change or "change something nearby".
|
||||
</critical>
|
||||
|
||||
<instruction>
|
||||
@@ -18,7 +18,7 @@ Line-addressed edits using hash-verified line references. Read file with hashes
|
||||
2. Collect the exact `LINE:HASH` refs you need
|
||||
3. Submit one `edit` call with all known operations for that file
|
||||
4. If another change on same file is needed later: re-read first, then edit
|
||||
5. Internally verify direction before submitting (`before token/expression` → `after token/expression`). Do not output prose; submit only the tool call.
|
||||
5. Direction-lock each operation before submitting (`exact source token/expression on target line` → `intended replacement`) and keep the mutation to one logical locus. Do not output prose; submit only the tool call.
|
||||
**Edit variants:**
|
||||
- `{ replaceLine: { loc: "LINE:HASH", content: "..." } }`
|
||||
- `{ replaceLines: { start: "LINE:HASH", end: "LINE:HASH", content: "..." } }`
|
||||
@@ -35,21 +35,27 @@ Line-addressed edits using hash-verified line references. Read file with hashes
|
||||
- Use `replaceLines` over a wide range when multiple `replaceLine` ops would work — wide ranges tempt reformatting everything in between
|
||||
|
||||
If a change spans multiple non-adjacent lines, use separate `replaceLine` operations for each — not a single `replaceLines` that includes unchanged lines in `content`.
|
||||
- Each edit operation must target a single logical change site. If a fix requires changes at two separate locations, use two separate edit operations — never a single `replaceLines` spanning both.
|
||||
- Self-check before submitting: if your edit would touch lines unrelated to the stated fix, split or narrow it.
|
||||
- Each edit operation must target one logical change site with minimal scope. If a fix requires two locations, use two operations; never span unrelated lines in one `replaceLines`.
|
||||
- Self-check before submitting: if your edit touches lines unrelated to the stated fix, split or narrow it.
|
||||
</caution>
|
||||
<instruction>
|
||||
**Recovery:**
|
||||
- Hash mismatch (`>>>` error): copy the updated `LINE:HASH` refs from the error verbatim and retry. Do NOT re-read the file unless you need lines not shown in the error.
|
||||
- Hash mismatch (`>>>` error): copy the updated `LINE:HASH` refs from the error verbatim and retry with the same intended mutation. Do NOT re-read unless you need lines not shown in the error.
|
||||
- If hash mismatch repeats after applying updated refs, stop blind retries and re-read the relevant region before retrying.
|
||||
- After a successful edit, always re-read the file before making another edit to the same file (hashes have changed).
|
||||
- No-op error ("identical content"): your replacement content matches what's already in the file. Re-read the target lines — the mutation is likely on a different line or the content has already been fixed.
|
||||
- No-op error ("identical content"): do not resend the same payload. Re-read the target lines, confirm mutation direction, and change either target refs or replacement content before retrying.
|
||||
</instruction>
|
||||
|
||||
<instruction>
|
||||
**Before submitting each edit call, verify:**
|
||||
- `path` is set and points to the correct file
|
||||
- Each `loc`/`start`/`end` ref matches `^\d+:[A-Za-z0-9]+$` — no spaces, no content after hash
|
||||
- `content` reproduces the original line's formatting with only the targeted change applied
|
||||
**Preflight schema and validation (required):**
|
||||
- Payload shape is `{"path": string, "edits": [operation, ...]}` with a non-empty `edits` array.
|
||||
- Each operation contains exactly one variant key: `replaceLine`, `replaceLines`, or `insertAfter`.
|
||||
- Required fields by variant:
|
||||
- `replaceLine`: `loc`, `content`
|
||||
- `replaceLines`: `start`, `end`, `content`
|
||||
- `insertAfter`: `loc`, `content` (non-empty)
|
||||
- Each `loc`/`start`/`end` ref matches `^\d+:[A-Za-z0-9]+$` (no spaces, no trailing source text).
|
||||
- `content` preserves original formatting and changes only the direction-locked target locus.
|
||||
</instruction>
|
||||
|
||||
<input>
|
||||
|
||||
@@ -60,17 +60,17 @@ export interface FormatResult {
|
||||
export async function formatContent(filePath: string, content: string): Promise<FormatResult> {
|
||||
const parser = parserByExtension[path.extname(filePath).toLowerCase()];
|
||||
if (!parser) {
|
||||
return { formatted: content, didFormat: false };
|
||||
}
|
||||
|
||||
try {
|
||||
const formatted = await prettier.format(content, { ...PRETTIER_OPTIONS, parser });
|
||||
return { formatted, didFormat: true };
|
||||
} catch {
|
||||
return { formatted: content, didFormat: false };
|
||||
}
|
||||
return { formatted: content, didFormat: false };
|
||||
}
|
||||
|
||||
try {
|
||||
const formatted = await prettier.format(content, { ...PRETTIER_OPTIONS, parser });
|
||||
return { formatted, didFormat: true };
|
||||
} catch {
|
||||
return { formatted: content, didFormat: false };
|
||||
}
|
||||
}
|
||||
|
||||
export async function formatDirectory(rootDir: string): Promise<void> {
|
||||
const files = await listFiles(rootDir);
|
||||
|
||||
|
||||
@@ -134,7 +134,7 @@ function isExcluded(filePath: string): boolean {
|
||||
}
|
||||
|
||||
function hasStructure(content: string): boolean {
|
||||
return ["function ", "class ", "export ", "=>"].some((token) => content.includes(token));
|
||||
return ["function ", "class ", "export ", "=>"].some(token => content.includes(token));
|
||||
}
|
||||
|
||||
async function collectFiles(reactDir: string): Promise<string[]> {
|
||||
@@ -149,7 +149,7 @@ async function collectFiles(reactDir: string): Promise<string[]> {
|
||||
if (entry.isDirectory()) {
|
||||
await walk(fullPath);
|
||||
} else if (entry.isFile()) {
|
||||
const ext = "." + entry.name.split(".").pop();
|
||||
const ext = `.${entry.name.split(".").pop()}`;
|
||||
if (SUPPORTED_EXTENSIONS.has(ext) && !isExcluded(fullPath)) {
|
||||
candidates.push(fullPath);
|
||||
}
|
||||
@@ -353,13 +353,13 @@ function findContainingFunction(entry: FileEntry, lineNumber: number): string |
|
||||
function getCandidatesForDifficulty(files: FileEntry[], difficulty: Difficulty): FileEntry[] {
|
||||
switch (difficulty) {
|
||||
case "easy":
|
||||
return files.filter((f) => f.lineCount < 150 && f.repeatedLines.size < 3);
|
||||
return files.filter(f => f.lineCount < 150 && f.repeatedLines.size < 3);
|
||||
case "medium":
|
||||
return files.filter((f) => f.lineCount >= 100 && f.lineCount <= 300);
|
||||
return files.filter(f => f.lineCount >= 100 && f.lineCount <= 300);
|
||||
case "hard":
|
||||
return files.filter((f) => f.lineCount > 200 && f.similarBlockCount >= 3);
|
||||
return files.filter(f => f.lineCount > 200 && f.similarBlockCount >= 3);
|
||||
case "nightmare":
|
||||
return files.filter((f) => f.repeatedLines.size > 0 && f.lineCount > 200);
|
||||
return files.filter(f => f.repeatedLines.size > 0 && f.lineCount > 200);
|
||||
default:
|
||||
return files;
|
||||
}
|
||||
@@ -388,7 +388,7 @@ async function isParsable(content: string, suffix: string): Promise<boolean> {
|
||||
|
||||
function regionAvailable(usedLines: Map<string, number[]>, filePath: string, lineNumber: number): boolean {
|
||||
const used = usedLines.get(filePath) ?? [];
|
||||
return !used.some((ln) => Math.abs(ln - lineNumber) <= 3);
|
||||
return !used.some(ln => Math.abs(ln - lineNumber) <= 3);
|
||||
}
|
||||
|
||||
function recordRegion(usedLines: Map<string, number[]>, filePath: string, lineNumber: number): void {
|
||||
@@ -486,9 +486,9 @@ async function generateCase(
|
||||
let candidates = getCandidatesForDifficulty(files, difficulty);
|
||||
if (candidates.length === 0) candidates = files;
|
||||
|
||||
let applicable = candidates.filter((entry) => mutation.canApply(entry.content));
|
||||
let applicable = candidates.filter(entry => mutation.canApply(entry.content));
|
||||
if (applicable.length === 0 && candidates !== files) {
|
||||
applicable = files.filter((entry) => mutation.canApply(entry.content));
|
||||
applicable = files.filter(entry => mutation.canApply(entry.content));
|
||||
}
|
||||
if (applicable.length === 0) return null;
|
||||
|
||||
@@ -500,7 +500,7 @@ async function generateCase(
|
||||
if (mutatedContent === entry.content) continue;
|
||||
if (!regionAvailable(usedLines, entry.path, info.lineNumber)) continue;
|
||||
|
||||
const suffix = "." + entry.path.split(".").pop();
|
||||
const suffix = `.${entry.path.split(".").pop()}`;
|
||||
if (!(await isParsable(mutatedContent, suffix))) continue;
|
||||
|
||||
const diffScore = scoreDifficulty(entry, info.lineNumber);
|
||||
@@ -620,14 +620,14 @@ async function main(): Promise<number> {
|
||||
}
|
||||
|
||||
console.log(`Analyzed ${files.length} files`);
|
||||
const withRepeats = files.filter((f) => f.repeatedLines.size > 0).length;
|
||||
const longFiles = files.filter((f) => f.lineCount > 200).length;
|
||||
const similarBlocks = files.filter((f) => f.similarBlockCount >= 3).length;
|
||||
const withRepeats = files.filter(f => f.repeatedLines.size > 0).length;
|
||||
const longFiles = files.filter(f => f.lineCount > 200).length;
|
||||
const similarBlocks = files.filter(f => f.similarBlockCount >= 3).length;
|
||||
console.log(` ${withRepeats} with repeated lines, ${longFiles} long files, ${similarBlocks} with similar blocks`);
|
||||
|
||||
const difficulties = args.difficulty
|
||||
.split(",")
|
||||
.map((s) => s.trim())
|
||||
.map(s => s.trim())
|
||||
.filter(Boolean) as Difficulty[];
|
||||
if (difficulties.length === 0) {
|
||||
console.error("No difficulties specified.");
|
||||
@@ -639,15 +639,15 @@ async function main(): Promise<number> {
|
||||
const categories = new Set(
|
||||
args.categories
|
||||
.split(",")
|
||||
.map((s) => s.trim())
|
||||
.map(s => s.trim())
|
||||
.filter(Boolean),
|
||||
);
|
||||
const unknown = Array.from(categories).filter((c) => !(c in CATEGORY_MAP));
|
||||
const unknown = Array.from(categories).filter(c => !(c in CATEGORY_MAP));
|
||||
if (unknown.length > 0) {
|
||||
console.error(`Unknown categories: ${unknown.join(", ")}`);
|
||||
return 1;
|
||||
}
|
||||
mutations = ALL_MUTATIONS.filter((m) => categories.has(m.category));
|
||||
mutations = ALL_MUTATIONS.filter(m => categories.has(m.category));
|
||||
}
|
||||
|
||||
const usedLines = new Map<string, number[]>();
|
||||
@@ -715,7 +715,7 @@ async function main(): Promise<number> {
|
||||
}
|
||||
}
|
||||
|
||||
const scores = results.map((r) => r.difficultyScore);
|
||||
const scores = results.map(r => r.difficultyScore);
|
||||
const min = Math.min(...scores);
|
||||
const max = Math.max(...scores);
|
||||
const avg = scores.reduce((a, b) => a + b, 0) / scores.length;
|
||||
@@ -731,7 +731,7 @@ async function main(): Promise<number> {
|
||||
await writeTarball(tarEntries, args.output);
|
||||
|
||||
console.log(`Generated ${results.length} cases in ${args.output}`);
|
||||
const scores = results.map((r) => r.difficultyScore);
|
||||
const scores = results.map(r => r.difficultyScore);
|
||||
const min = Math.min(...scores);
|
||||
const max = Math.max(...scores);
|
||||
const avg = scores.reduce((a, b) => a + b, 0) / scores.length;
|
||||
|
||||
@@ -33,7 +33,7 @@ function generateReportFilename(config: BenchmarkConfig, format: "markdown" | "j
|
||||
|
||||
function printUsage(tasks?: EditTask[]): void {
|
||||
const taskList = tasks
|
||||
? tasks.map((t) => ` ${t.id.padEnd(30)} ${t.name}`).join("\n")
|
||||
? tasks.map(t => ` ${t.id.padEnd(30)} ${t.name}`).join("\n")
|
||||
: " (use --list to see available tasks)";
|
||||
console.log(`
|
||||
Edit Benchmark - Evaluate patch application success rates
|
||||
@@ -59,6 +59,8 @@ Options:
|
||||
--guided Include an authoritative suggested edit payload (default: false)
|
||||
--no-guided Disable guided mode
|
||||
--max-attempts <n> Max prompt attempts per run (default: 1)
|
||||
--no-op-retry-limit <n> Stop after repeated preventable no-op failures (default: 2)
|
||||
--mutation-scope-window <n> Allowed line-distance from mutation target for hashline refs (default: 20)
|
||||
--output <file> Output file (default: run_<model>_<variant>_<fuzzy>_<threshold>_<timestamp>.md)
|
||||
--format <fmt> Output format: markdown, json (default: markdown)
|
||||
--check-fixtures Validate fixtures and exit
|
||||
@@ -92,8 +94,8 @@ Examples:
|
||||
|
||||
async function resolveExtractedDir(tempDir: string): Promise<string> {
|
||||
const entries = await fs.promises.readdir(tempDir, { withFileTypes: true });
|
||||
const dirs = entries.filter((entry) => entry.isDirectory());
|
||||
const files = entries.filter((entry) => entry.isFile());
|
||||
const dirs = entries.filter(entry => entry.isDirectory());
|
||||
const files = entries.filter(entry => entry.isFile());
|
||||
if (dirs.length === 1 && files.length === 0) {
|
||||
return join(tempDir, dirs[0]!.name);
|
||||
}
|
||||
@@ -155,6 +157,8 @@ async function main(): Promise<void> {
|
||||
guided: { type: "boolean", default: false },
|
||||
"no-guided": { type: "boolean", default: false },
|
||||
"max-attempts": { type: "string", default: "1" },
|
||||
"no-op-retry-limit": { type: "string", default: "2" },
|
||||
"mutation-scope-window": { type: "string", default: "20" },
|
||||
"timeout-retries": { type: "string", default: "1" },
|
||||
"require-edit-tool-call": { type: "boolean", default: false },
|
||||
"require-read-tool-call": { type: "boolean", default: false },
|
||||
@@ -218,44 +222,44 @@ async function main(): Promise<void> {
|
||||
}
|
||||
|
||||
const runsPerTask = parseInt(values.runs!, 10);
|
||||
if (isNaN(runsPerTask) || runsPerTask < 1) {
|
||||
if (Number.isNaN(runsPerTask) || runsPerTask < 1) {
|
||||
console.error(`Invalid runs value: ${values.runs}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const timeout = parseInt(values.timeout!, 10);
|
||||
if (isNaN(timeout) || timeout < 1000) {
|
||||
if (Number.isNaN(timeout) || timeout < 1000) {
|
||||
console.error(`Invalid timeout value: ${values.timeout}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const taskConcurrency = parseInt(values["task-concurrency"]!, 10);
|
||||
if (isNaN(taskConcurrency) || taskConcurrency < 1) {
|
||||
if (Number.isNaN(taskConcurrency) || taskConcurrency < 1) {
|
||||
console.error(`Invalid task concurrency value: ${values["task-concurrency"]}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const maxAttempts = parseInt(values["max-attempts"] ?? "2", 10);
|
||||
if (isNaN(maxAttempts) || maxAttempts < 1 || maxAttempts > 5) {
|
||||
console.error(`Invalid max-attempts value: ${values["max-attempts"]}. Must be 1-5.`);
|
||||
if (Number.isNaN(maxAttempts) || maxAttempts < 1 || maxAttempts > 5) {
|
||||
console.error(`Invalid max-attempts value: ${values["max-attempts"]}. Must be 1-5.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const timeoutRetryCount = parseInt(values["timeout-retries"] ?? "1", 10);
|
||||
if (isNaN(timeoutRetryCount) || timeoutRetryCount < 0 || timeoutRetryCount > 3) {
|
||||
if (Number.isNaN(timeoutRetryCount) || timeoutRetryCount < 0 || timeoutRetryCount > 3) {
|
||||
console.error(`Invalid timeout-retries value: ${values["timeout-retries"]}. Must be 0-3.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
let tasksToRun = allTasks;
|
||||
if (values.tasks) {
|
||||
const taskIds = values.tasks.split(",").map((s) => s.trim());
|
||||
const taskIds = values.tasks.split(",").map(s => s.trim());
|
||||
tasksToRun = [];
|
||||
for (const id of taskIds) {
|
||||
const task = allTasks.find((t) => t.id === id);
|
||||
const task = allTasks.find(t => t.id === id);
|
||||
if (!task) {
|
||||
console.error(`Unknown task ID: ${id}`);
|
||||
console.error(`Available tasks: ${allTasks.map((t) => t.id).join(", ")}`);
|
||||
console.error(`Available tasks: ${allTasks.map(t => t.id).join(", ")}`);
|
||||
process.exit(1);
|
||||
}
|
||||
tasksToRun.push(task);
|
||||
@@ -297,7 +301,7 @@ async function main(): Promise<void> {
|
||||
editFuzzyThreshold = "auto";
|
||||
} else {
|
||||
const parsed = parseFloat(values["edit-fuzzy-threshold"]);
|
||||
if (isNaN(parsed) || parsed < 0 || parsed > 1) {
|
||||
if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) {
|
||||
console.error(`Invalid edit-fuzzy-threshold: ${values["edit-fuzzy-threshold"]}. Must be 0-1 or auto.`);
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -315,11 +319,11 @@ async function main(): Promise<void> {
|
||||
timeout,
|
||||
taskConcurrency,
|
||||
autoFormat: values["auto-format"],
|
||||
guided,
|
||||
maxAttempts,
|
||||
guided,
|
||||
maxAttempts,
|
||||
timeoutRetryCount,
|
||||
requireEditToolCall: values["require-edit-tool-call"],
|
||||
requireReadToolCall: values["require-read-tool-call"],
|
||||
requireReadToolCall: values["require-read-tool-call"],
|
||||
noEditRequired: values["no-edit-required"],
|
||||
editVariant,
|
||||
editFuzzy,
|
||||
@@ -346,7 +350,7 @@ async function main(): Promise<void> {
|
||||
console.log("Require edit tool call: yes");
|
||||
}
|
||||
if (config.requireReadToolCall) {
|
||||
console.log("Require read tool call: yes");
|
||||
console.log("Require read tool call: yes");
|
||||
}
|
||||
if (config.noEditRequired) {
|
||||
console.log("No-edit-required baseline: yes");
|
||||
@@ -364,7 +368,7 @@ async function main(): Promise<void> {
|
||||
console.log("");
|
||||
|
||||
const progress = new LiveProgress(tasksToRun.length * config.runsPerTask, config.runsPerTask);
|
||||
const result = await runBenchmark(tasksToRun, config, (event) => {
|
||||
const result = await runBenchmark(tasksToRun, config, event => {
|
||||
progress.handleEvent(event);
|
||||
});
|
||||
progress.finish();
|
||||
@@ -441,7 +445,9 @@ class LiveProgress {
|
||||
|
||||
if (event.result && !event.result.success && event.result.error) {
|
||||
this.#flushLine();
|
||||
console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed: ${event.result.error}`);
|
||||
console.log(
|
||||
` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed: ${event.result.error}`,
|
||||
);
|
||||
if (event.result.diff) {
|
||||
const diffLines = event.result.diff.split("\n").slice(0, 30);
|
||||
if (diffLines.length > 0) {
|
||||
@@ -481,7 +487,9 @@ class LiveProgress {
|
||||
console.log("");
|
||||
console.log("Runtime Stats:");
|
||||
console.log(` Task success: ${successRate.toFixed(1)}% (${this.#success}/${n})`);
|
||||
console.log(` Edit success: ${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`);
|
||||
console.log(
|
||||
` Edit success: ${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`,
|
||||
);
|
||||
console.log(` Avg indent score: ${avgIndent.toFixed(2)}`);
|
||||
console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
|
||||
console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
|
||||
@@ -524,13 +532,13 @@ class LiveProgress {
|
||||
return;
|
||||
}
|
||||
if (this.#lastLineLength > 0) {
|
||||
process.stdout.write("\r" + padding(this.#lastLineLength) + "\r");
|
||||
process.stdout.write(`\r${padding(this.#lastLineLength)}\r`);
|
||||
this.#lastLineLength = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
main().catch(err => {
|
||||
console.error("Benchmark failed:", err);
|
||||
process.exit(1);
|
||||
});
|
||||
|
||||
@@ -37,6 +37,11 @@ function isCommented(line: string, index: number): boolean {
|
||||
return commentIndex !== -1 && commentIndex < index;
|
||||
}
|
||||
|
||||
function* execAll(regex: RegExp, text: string): Generator<RegExpExecArray> {
|
||||
const re = new RegExp(regex.source, regex.flags.includes("g") ? regex.flags : `${regex.flags}g`);
|
||||
for (let m = re.exec(text); m !== null; m = re.exec(text)) yield m;
|
||||
}
|
||||
|
||||
function iterCandidates(
|
||||
lines: string[],
|
||||
pattern: RegExp,
|
||||
@@ -45,9 +50,7 @@ function iterCandidates(
|
||||
const candidates: Candidate[] = [];
|
||||
for (let lineNumber = 1; lineNumber <= lines.length; lineNumber++) {
|
||||
const line = lines[lineNumber - 1];
|
||||
const regex = new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : pattern.flags + "g");
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(line)) !== null) {
|
||||
for (const match of execAll(pattern, line)) {
|
||||
if (isCommented(line, match.index)) continue;
|
||||
const replacement = replacementFn(match);
|
||||
if (replacement === null) continue;
|
||||
@@ -86,7 +89,7 @@ function applyCandidate(lines: string[], candidate: Candidate): MutationInfo {
|
||||
|
||||
function stripStrings(line: string): string {
|
||||
const pattern = /(?<quote>['"])(?<body>(?:\\.|[^\\\n])*?)\k<quote>/g;
|
||||
return line.replace(pattern, (match) => padding(match.length));
|
||||
return line.replace(pattern, match => padding(match.length));
|
||||
}
|
||||
|
||||
function mutateIdentifier(identifier: string): string | null {
|
||||
@@ -144,7 +147,7 @@ class SwapComparisonMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -166,7 +169,7 @@ class SwapEqualityMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -188,7 +191,7 @@ class SwapLogicalMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -231,7 +234,7 @@ class SwapIncDecMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -253,7 +256,7 @@ class SwapArithmeticMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -275,7 +278,7 @@ class BooleanLiteralFlipMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m[1]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m[1]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -289,7 +292,7 @@ class OptionalChainRemovalMutation extends BaseMutation {
|
||||
"Restore the optional chaining operator (`?.`) at the ONE location where it was removed. Do not add optional chaining elsewhere.";
|
||||
description = "Optional chaining was removed from a property access.";
|
||||
|
||||
#pattern = /\?\.(?=[\w\[(])/;
|
||||
#pattern = /\?\.(?=[\w[(])/;
|
||||
|
||||
canApply(content: string): boolean {
|
||||
return this.#pattern.test(content);
|
||||
@@ -321,7 +324,7 @@ class CallArgumentSwapMutation extends BaseMutation {
|
||||
const candidate = pickCandidate(
|
||||
lines,
|
||||
this.#pattern,
|
||||
(m) => {
|
||||
m => {
|
||||
const callee = m.groups?.callee ?? "";
|
||||
const a = m.groups?.a ?? "";
|
||||
const b = m.groups?.b ?? "";
|
||||
@@ -350,7 +353,7 @@ class NullishCoalescingSwapMutation extends BaseMutation {
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
const lines = content.split("\n");
|
||||
const candidate = pickCandidate(lines, this.#pattern, (m) => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
const candidate = pickCandidate(lines, this.#pattern, m => this.#swap[m.groups?.op ?? m[0]] ?? null, rng);
|
||||
if (!candidate) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
const info = applyCandidate(lines, candidate);
|
||||
return [lines.join("\n"), info];
|
||||
@@ -381,7 +384,7 @@ class RegexQuantifierSwapMutation extends BaseMutation {
|
||||
lineCounts.set(line, (lineCounts.get(line) ?? 0) + 1);
|
||||
}
|
||||
|
||||
const repeatedCandidates = candidates.filter((c) => (lineCounts.get(lines[c.lineNumber - 1]) ?? 0) > 1);
|
||||
const repeatedCandidates = candidates.filter(c => (lineCounts.get(lines[c.lineNumber - 1]) ?? 0) > 1);
|
||||
|
||||
const candidate = randomChoice(repeatedCandidates.length > 0 ? repeatedCandidates : candidates, rng);
|
||||
const info = applyCandidate(lines, candidate);
|
||||
@@ -392,9 +395,7 @@ class RegexQuantifierSwapMutation extends BaseMutation {
|
||||
const candidates: Candidate[] = [];
|
||||
for (let lineNumber = 1; lineNumber <= lines.length; lineNumber++) {
|
||||
const line = lines[lineNumber - 1];
|
||||
const litRegex = new RegExp(this.#literalPattern.source, this.#literalPattern.flags);
|
||||
let litMatch: RegExpExecArray | null;
|
||||
while ((litMatch = litRegex.exec(line)) !== null) {
|
||||
for (const litMatch of execAll(this.#literalPattern, line)) {
|
||||
if (isCommented(line, litMatch.index)) continue;
|
||||
const prefix = line.slice(0, litMatch.index);
|
||||
if (prefix && !" =({[,;:!".includes(prefix[prefix.length - 1]) && !/\s/.test(prefix[prefix.length - 1])) {
|
||||
@@ -402,9 +403,7 @@ class RegexQuantifierSwapMutation extends BaseMutation {
|
||||
}
|
||||
const bodyStart = litMatch.index + 1; // after opening /
|
||||
const body = litMatch.groups?.body ?? "";
|
||||
const quantRegex = new RegExp(this.#quantPattern.source, this.#quantPattern.flags);
|
||||
let tokenMatch: RegExpExecArray | null;
|
||||
while ((tokenMatch = quantRegex.exec(body)) !== null) {
|
||||
for (const tokenMatch of execAll(this.#quantPattern, body)) {
|
||||
const quantifier = tokenMatch.groups?.quant ?? tokenMatch[2];
|
||||
const swapped = quantifier === "+" ? "*" : "+";
|
||||
const start = bodyStart + tokenMatch.index + tokenMatch[0].length - 1;
|
||||
@@ -439,9 +438,7 @@ class UnicodeHyphenMutation extends BaseMutation {
|
||||
const candidates: Candidate[] = [];
|
||||
for (let lineNumber = 1; lineNumber <= lines.length; lineNumber++) {
|
||||
const line = lines[lineNumber - 1];
|
||||
const regex = new RegExp(this.#stringPattern.source, this.#stringPattern.flags);
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(line)) !== null) {
|
||||
for (const match of execAll(this.#stringPattern, line)) {
|
||||
if (isCommented(line, match.index)) continue;
|
||||
const body = match.groups?.body ?? "";
|
||||
const dashIndex = body.indexOf("-");
|
||||
@@ -530,9 +527,7 @@ class IdentifierMultiEditMutation extends BaseMutation {
|
||||
for (let lineNumber = 1; lineNumber <= lines.length; lineNumber++) {
|
||||
const line = lines[lineNumber - 1];
|
||||
const masked = stripStrings(line);
|
||||
const regex = new RegExp(this.#pattern.source, this.#pattern.flags);
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(masked)) !== null) {
|
||||
for (const match of execAll(this.#pattern, masked)) {
|
||||
if (isCommented(line, match.index)) continue;
|
||||
const identifier = match[0];
|
||||
if (this.#keywords.has(identifier)) continue;
|
||||
@@ -544,13 +539,9 @@ class IdentifierMultiEditMutation extends BaseMutation {
|
||||
}
|
||||
}
|
||||
|
||||
let candidates = Array.from(occurrences.entries()).filter(
|
||||
([, spans]) => new Set(spans.map((s) => s[0])).size >= 3,
|
||||
);
|
||||
let candidates = Array.from(occurrences.entries()).filter(([, spans]) => new Set(spans.map(s => s[0])).size >= 3);
|
||||
if (candidates.length === 0) {
|
||||
candidates = Array.from(occurrences.entries()).filter(
|
||||
([, spans]) => new Set(spans.map((s) => s[0])).size >= 2,
|
||||
);
|
||||
candidates = Array.from(occurrences.entries()).filter(([, spans]) => new Set(spans.map(s => s[0])).size >= 2);
|
||||
}
|
||||
if (candidates.length === 0) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
|
||||
@@ -558,12 +549,12 @@ class IdentifierMultiEditMutation extends BaseMutation {
|
||||
const mutated = mutateIdentifier(identifier);
|
||||
if (mutated === null) return [content, { lineNumber: 0, originalSnippet: "", mutatedSnippet: "" }];
|
||||
|
||||
const lineNumbers = Array.from(new Set(spans.map((s) => s[0])));
|
||||
const lineNumbers = Array.from(new Set(spans.map(s => s[0])));
|
||||
const editCount = Math.min(lineNumbers.length, randomChoice(lineNumbers.length >= 3 ? [2, 3, 3, 4] : [2], rng));
|
||||
const chosenLines = randomSample(lineNumbers, editCount, rng);
|
||||
const selectedSpans: Array<[number, number, number]> = [];
|
||||
for (const ln of chosenLines) {
|
||||
const lineSpans = spans.filter((s) => s[0] === ln);
|
||||
const lineSpans = spans.filter(s => s[0] === ln);
|
||||
selectedSpans.push(randomChoice(lineSpans, rng));
|
||||
}
|
||||
|
||||
@@ -618,9 +609,7 @@ class DuplicateLineLiteralFlipMutation extends BaseMutation {
|
||||
[this.#eqPattern, this.#eqSwap],
|
||||
[this.#compPattern, this.#compSwap],
|
||||
] as const) {
|
||||
const regex = new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : pattern.flags + "g");
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(line)) !== null) {
|
||||
for (const match of execAll(pattern, line)) {
|
||||
if (isCommented(line, match.index)) continue;
|
||||
const token = match[0];
|
||||
const replacement = swapMap[token];
|
||||
@@ -651,8 +640,7 @@ class SwapAdjacentLinesMutation extends BaseMutation {
|
||||
fixHint = "Swap the two adjacent lines back to their original order.";
|
||||
description = "Two adjacent statements are in the wrong order.";
|
||||
|
||||
#statementPattern =
|
||||
/^\s*(?:(?:const|let|var)\s+\w+\s*=|return\s+|\w+\s*(?:\.\w+)*\s*\(|\w+\s*(?:\.\w+)*\s*=)/;
|
||||
#statementPattern = /^\s*(?:(?:const|let|var)\s+\w+\s*=|return\s+|\w+\s*(?:\.\w+)*\s*\(|\w+\s*(?:\.\w+)*\s*=)/;
|
||||
|
||||
canApply(content: string): boolean {
|
||||
const lines = content.split("\n");
|
||||
@@ -708,7 +696,7 @@ class SwapIfElseBranchesMutation extends BaseMutation {
|
||||
|
||||
canApply(content: string): boolean {
|
||||
if (!content.includes("} else {")) return false;
|
||||
return content.split("\n").some((line) => this.#ifPattern.test(line));
|
||||
return content.split("\n").some(line => this.#ifPattern.test(line));
|
||||
}
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
@@ -821,7 +809,7 @@ class RemoveEarlyReturnMutation extends BaseMutation {
|
||||
newLines.join("\n"),
|
||||
{
|
||||
lineNumber: i + 1,
|
||||
originalSnippet: removedLines.map((l) => l.trim()).join("\n"),
|
||||
originalSnippet: removedLines.map(l => l.trim()).join("\n"),
|
||||
mutatedSnippet: "[removed]",
|
||||
},
|
||||
];
|
||||
@@ -838,15 +826,13 @@ class SwapNamedImportsMutation extends BaseMutation {
|
||||
#importPattern = /import\s*\{(?<imports>[^}]+)\}\s*from\s*['"]/;
|
||||
|
||||
canApply(content: string): boolean {
|
||||
const regex = new RegExp(this.#importPattern.source, "g");
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(content)) !== null) {
|
||||
for (const match of execAll(this.#importPattern, content)) {
|
||||
const imports = match.groups?.imports ?? "";
|
||||
const parts = imports
|
||||
.split(",")
|
||||
.map((p) => p.trim())
|
||||
.map(p => p.trim())
|
||||
.filter(Boolean);
|
||||
const simpleParts = parts.filter((p) => !p.includes(" as ") && /^\w+$/.test(p));
|
||||
const simpleParts = parts.filter(p => !p.includes(" as ") && /^\w+$/.test(p));
|
||||
if (simpleParts.length >= 2) return true;
|
||||
}
|
||||
return false;
|
||||
@@ -858,11 +844,9 @@ class SwapNamedImportsMutation extends BaseMutation {
|
||||
|
||||
for (let lineNumber = 1; lineNumber <= lines.length; lineNumber++) {
|
||||
const line = lines[lineNumber - 1];
|
||||
const regex = new RegExp(this.#importPattern.source, "g");
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(line)) !== null) {
|
||||
for (const match of execAll(this.#importPattern, line)) {
|
||||
const importsStr = match.groups?.imports ?? "";
|
||||
const parts = importsStr.split(",").map((p) => p.trim());
|
||||
const parts = importsStr.split(",").map(p => p.trim());
|
||||
const simpleIndices = parts
|
||||
.map((p, idx) => ({ p, idx }))
|
||||
.filter(({ p }) => p && !p.includes(" as ") && /^\w+$/.test(p))
|
||||
@@ -905,7 +889,7 @@ class DeleteStatementMutation extends BaseMutation {
|
||||
#statementPattern = /^\s*(?:(?:const|let|var)\s+\w+\s*=.+;|\w+\s*\+=.+;|\w+\s*-=.+;|\w+\s*=\s*\w+.+;)\s*$/;
|
||||
|
||||
canApply(content: string): boolean {
|
||||
return content.split("\n").some((line) => this.#statementPattern.test(line));
|
||||
return content.split("\n").some(line => this.#statementPattern.test(line));
|
||||
}
|
||||
|
||||
mutate(content: string, rng: () => number): [string, MutationInfo] {
|
||||
@@ -946,8 +930,8 @@ class OffByOneMutation extends BaseMutation {
|
||||
[/(?<=[\s(,=<>])1(?=[\s),;])/, () => "0"],
|
||||
[/\.length\s*-\s*1(?=[\s),;\]])/, () => ".length - 2"],
|
||||
[/\.length\s*-\s*2(?=[\s),;\]])/, () => ".length - 1"],
|
||||
[/<\s*(\w+\.length)/, (m) => `<= ${m[1]}`],
|
||||
[/<=\s*(\w+\.length)/, (m) => `< ${m[1]}`],
|
||||
[/<\s*(\w+\.length)/, m => `<= ${m[1]}`],
|
||||
[/<=\s*(\w+\.length)/, m => `< ${m[1]}`],
|
||||
];
|
||||
|
||||
canApply(content: string): boolean {
|
||||
@@ -972,9 +956,7 @@ class OffByOneMutation extends BaseMutation {
|
||||
}
|
||||
|
||||
for (const [pattern, replacementFn] of this.#patterns) {
|
||||
const regex = new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : pattern.flags + "g");
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = regex.exec(line)) !== null) {
|
||||
for (const match of execAll(pattern, line)) {
|
||||
if (isCommented(line, match.index)) continue;
|
||||
const original = match[0];
|
||||
const replacement = replacementFn(match);
|
||||
@@ -1021,14 +1003,14 @@ export const ALL_MUTATIONS: Mutation[] = [
|
||||
];
|
||||
|
||||
export const CATEGORY_MAP: Record<string, string[]> = {
|
||||
operator: ALL_MUTATIONS.filter((m) => m.category === "operator").map((m) => m.name),
|
||||
literal: ALL_MUTATIONS.filter((m) => m.category === "literal").map((m) => m.name),
|
||||
access: ALL_MUTATIONS.filter((m) => m.category === "access").map((m) => m.name),
|
||||
call: ALL_MUTATIONS.filter((m) => m.category === "call").map((m) => m.name),
|
||||
regex: ALL_MUTATIONS.filter((m) => m.category === "regex").map((m) => m.name),
|
||||
unicode: ALL_MUTATIONS.filter((m) => m.category === "unicode").map((m) => m.name),
|
||||
identifier: ALL_MUTATIONS.filter((m) => m.category === "identifier").map((m) => m.name),
|
||||
duplicate: ALL_MUTATIONS.filter((m) => m.category === "duplicate").map((m) => m.name),
|
||||
structural: ALL_MUTATIONS.filter((m) => m.category === "structural").map((m) => m.name),
|
||||
import: ALL_MUTATIONS.filter((m) => m.category === "import").map((m) => m.name),
|
||||
operator: ALL_MUTATIONS.filter(m => m.category === "operator").map(m => m.name),
|
||||
literal: ALL_MUTATIONS.filter(m => m.category === "literal").map(m => m.name),
|
||||
access: ALL_MUTATIONS.filter(m => m.category === "access").map(m => m.name),
|
||||
call: ALL_MUTATIONS.filter(m => m.category === "call").map(m => m.name),
|
||||
regex: ALL_MUTATIONS.filter(m => m.category === "regex").map(m => m.name),
|
||||
unicode: ALL_MUTATIONS.filter(m => m.category === "unicode").map(m => m.name),
|
||||
identifier: ALL_MUTATIONS.filter(m => m.category === "identifier").map(m => m.name),
|
||||
duplicate: ALL_MUTATIONS.filter(m => m.category === "duplicate").map(m => m.name),
|
||||
structural: ALL_MUTATIONS.filter(m => m.category === "structural").map(m => m.name),
|
||||
import: ALL_MUTATIONS.filter(m => m.category === "import").map(m => m.name),
|
||||
};
|
||||
|
||||
@@ -42,10 +42,10 @@ function escapeMarkdown(text: string): string {
|
||||
|
||||
function truncateText(text: string, maxLength: number): string {
|
||||
if (text.length <= maxLength) return text;
|
||||
return text.slice(0, maxLength - 3) + "...";
|
||||
return `${text.slice(0, maxLength - 3)}...`;
|
||||
}
|
||||
|
||||
function formatEditArgs(args: unknown, maxLength: number): string {
|
||||
function _formatEditArgs(args: unknown, maxLength: number): string {
|
||||
if (!args || typeof args !== "object") return "—";
|
||||
const diff = (args as { diff?: unknown }).diff;
|
||||
if (typeof diff === "string") {
|
||||
@@ -88,10 +88,10 @@ function formatFiles(files: string[]): string {
|
||||
export function generateReport(result: BenchmarkResult): string {
|
||||
const { config, tasks, summary } = result;
|
||||
const runsPerTask = config.runsPerTask;
|
||||
const allRuns = tasks.flatMap((task) => task.runs);
|
||||
const verifiedRuns = allRuns.filter((run) => run.verificationPassed).length;
|
||||
const editToolRuns = allRuns.filter((run) => run.patchApplied).length;
|
||||
const successRuns = allRuns.filter((run) => run.success).length;
|
||||
const allRuns = tasks.flatMap(task => task.runs);
|
||||
const verifiedRuns = allRuns.filter(run => run.verificationPassed).length;
|
||||
const editToolRuns = allRuns.filter(run => run.patchApplied).length;
|
||||
const successRuns = allRuns.filter(run => run.success).length;
|
||||
const totalEditAttempts = allRuns.reduce((sum, run) => sum + run.toolCalls.edit, 0);
|
||||
const totalEditFailures = allRuns.reduce((sum, run) => sum + run.toolCalls.editFailures, 0);
|
||||
|
||||
@@ -114,6 +114,9 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
);
|
||||
lines.push(`| Guided Mode | ${config.guided === false ? "no" : "yes"} |`);
|
||||
lines.push(`| Max Attempts | ${config.maxAttempts ?? 1} |`);
|
||||
lines.push(`| Timeout Retries | ${config.timeoutRetryCount ?? 1} |`);
|
||||
lines.push(`| No-op Retry Limit | ${config.noOpRetryLimit ?? 2} |`);
|
||||
lines.push(`| Mutation Scope Window | ${config.mutationScopeWindow ?? 20} |`);
|
||||
lines.push(`| Require Edit Tool | ${config.requireEditToolCall ? "yes" : "no"} |`);
|
||||
lines.push(`| Require Read Tool | ${config.requireReadToolCall ? "yes" : "no"} |`);
|
||||
lines.push(`| No-Edit Baseline | ${config.noEditRequired ? "yes" : "no"} |`);
|
||||
@@ -134,6 +137,17 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
if (typeof summary.mutationIntentMatchRate === "number") {
|
||||
lines.push(`| Mutation Intent Match Rate | ${formatPercent(summary.mutationIntentMatchRate)} |`);
|
||||
}
|
||||
const preventableCounts = summary.preventableFailureCounts ?? {};
|
||||
const preventableTotal = Object.values(preventableCounts).reduce((sum, count) => sum + (count ?? 0), 0);
|
||||
if (preventableTotal > 0) {
|
||||
lines.push(`| Preventable Diagnostics | ${preventableTotal} |`);
|
||||
for (const kind of ["malformed_edit_payload", "stale_hash", "off_target_edit", "no_op_retry_loop", "timeout"]) {
|
||||
const count = preventableCounts[kind as keyof typeof preventableCounts];
|
||||
if (typeof count === "number" && count > 0) {
|
||||
lines.push(`| - ${kind} | ${count} |`);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (config.editVariant === "patch" || config.editVariant === "hashline") {
|
||||
lines.push(`| Patch Failure Rate | ${formatRate(totalEditFailures, totalEditAttempts)} |`);
|
||||
}
|
||||
@@ -213,7 +227,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("");
|
||||
|
||||
for (const task of tasks) {
|
||||
const taskFailures = task.runs.filter((run) => run.editFailures.length > 0);
|
||||
const taskFailures = task.runs.filter(run => run.editFailures.length > 0);
|
||||
if (taskFailures.length === 0) continue;
|
||||
lines.push(`### ${task.name} (${formatFiles(task.files)})`);
|
||||
lines.push("");
|
||||
@@ -248,7 +262,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
}
|
||||
}
|
||||
|
||||
const flakyTasks = tasks.filter((t) => t.successRate > 0 && t.successRate < 1);
|
||||
const flakyTasks = tasks.filter(t => t.successRate > 0 && t.successRate < 1);
|
||||
if (flakyTasks.length > 0) {
|
||||
lines.push("## Flaky Tasks (partial passing)");
|
||||
lines.push("");
|
||||
@@ -271,7 +285,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
}
|
||||
}
|
||||
|
||||
const failedTasks = tasks.filter((t) => t.successRate === 0);
|
||||
const failedTasks = tasks.filter(t => t.successRate === 0);
|
||||
if (failedTasks.length > 0) {
|
||||
lines.push("## Failed Tasks (0% passing)");
|
||||
lines.push("");
|
||||
@@ -280,7 +294,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push(`### ${task.name} (${formatFiles(task.files)}) — 0/${runsPerTask}`);
|
||||
lines.push("");
|
||||
|
||||
const errors = task.runs.map((r) => r.error).filter(Boolean);
|
||||
const errors = task.runs.map(r => r.error).filter(Boolean);
|
||||
const uniqueErrors = [...new Set(errors)];
|
||||
|
||||
if (uniqueErrors.length === 1) {
|
||||
@@ -305,7 +319,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("");
|
||||
}
|
||||
|
||||
const sampleResponse = task.runs.find((r) => r.agentResponse)?.agentResponse;
|
||||
const sampleResponse = task.runs.find(r => r.agentResponse)?.agentResponse;
|
||||
if (sampleResponse) {
|
||||
lines.push("**Sample agent response (run 1):**");
|
||||
lines.push("```");
|
||||
@@ -314,7 +328,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("");
|
||||
}
|
||||
|
||||
const sampleDiff = task.runs.find((r) => r.diff)?.diff;
|
||||
const sampleDiff = task.runs.find(r => r.diff)?.diff;
|
||||
if (sampleDiff) {
|
||||
lines.push("**Diff (expected vs actual):**");
|
||||
lines.push("```diff");
|
||||
@@ -332,7 +346,7 @@ function formatScore(value: number): string {
|
||||
return value.toFixed(2);
|
||||
}
|
||||
|
||||
function findTaskPrompt(task: TaskResult): { prompt: string } | undefined {
|
||||
function findTaskPrompt(_task: TaskResult): { prompt: string } | undefined {
|
||||
// This is a placeholder - in actual use, we'd pass task definitions alongside results
|
||||
return undefined;
|
||||
}
|
||||
@@ -342,12 +356,12 @@ export function generateJsonReport(result: BenchmarkResult): string {
|
||||
}
|
||||
|
||||
function appendCategorySummary(lines: string[], tasks: TaskResult[]): void {
|
||||
const runs = tasks.flatMap((task) => task.runs);
|
||||
const runs = tasks.flatMap(task => task.runs);
|
||||
const categoryStats = new Map<string, { runs: number; verified: number; editUsed: number; success: number }>();
|
||||
const difficultyByCategory = new Map<string, number[]>();
|
||||
|
||||
for (const task of tasks) {
|
||||
const metadata = task.runs.find((run) => run.mutationCategory || run.difficultyScore !== undefined);
|
||||
const metadata = task.runs.find(run => run.mutationCategory || run.difficultyScore !== undefined);
|
||||
const category = metadata?.mutationCategory ?? "unknown";
|
||||
if (typeof metadata?.difficultyScore === "number") {
|
||||
const scores = difficultyByCategory.get(category) ?? [];
|
||||
@@ -383,7 +397,7 @@ function appendCategorySummary(lines: string[], tasks: TaskResult[]): void {
|
||||
}
|
||||
|
||||
function appendMutationSummary(lines: string[], tasks: TaskResult[]): void {
|
||||
const runs = tasks.flatMap((task) => task.runs);
|
||||
const runs = tasks.flatMap(task => task.runs);
|
||||
const mutationStats = new Map<
|
||||
string,
|
||||
{ category: string; runs: number; verified: number; editUsed: number; success: number }
|
||||
@@ -422,7 +436,7 @@ function appendMutationSummary(lines: string[], tasks: TaskResult[]): void {
|
||||
}
|
||||
|
||||
function appendDifficultySummary(lines: string[], tasks: TaskResult[]): void {
|
||||
const runs = tasks.flatMap((task) => task.runs);
|
||||
const runs = tasks.flatMap(task => task.runs);
|
||||
const buckets = [
|
||||
{ label: "0-2", min: 0, max: 2 },
|
||||
{ label: "3-5", min: 3, max: 5 },
|
||||
@@ -441,7 +455,7 @@ function appendDifficultySummary(lines: string[], tasks: TaskResult[]): void {
|
||||
if (run.success) unknown.success += 1;
|
||||
continue;
|
||||
}
|
||||
const bucket = buckets.find((entry) => score >= entry.min && score <= entry.max);
|
||||
const bucket = buckets.find(entry => score >= entry.min && score <= entry.max);
|
||||
const label = bucket?.label ?? "unknown";
|
||||
const entry = bucketStats.get(label) ?? { runs: 0, verified: 0, editUsed: 0, success: 0 };
|
||||
entry.runs += 1;
|
||||
|
||||
@@ -10,15 +10,14 @@ import { join } from "node:path";
|
||||
import type { ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import { RpcClient } from "@oh-my-pi/pi-coding-agent";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { diffLines } from "diff";
|
||||
import { renderPromptTemplate } from "../coding-agent/src/config/prompt-templates";
|
||||
import { computeLineHash } from "../coding-agent/src/patch/hashline";
|
||||
import { diffLines } from "diff";
|
||||
import { formatDirectory } from "./formatter";
|
||||
import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" };
|
||||
import { type EditTask, extractTaskFiles } from "./tasks";
|
||||
import { verifyExpectedFileSubset, verifyExpectedFiles } from "./verify";
|
||||
|
||||
import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" };
|
||||
|
||||
const TMP_DIR = await TempDir.create("@reach-benchmark-");
|
||||
const TMP = TMP_DIR.path();
|
||||
|
||||
@@ -54,7 +53,7 @@ function getEditPathFromArgs(args: unknown): string | null {
|
||||
const HASHLINE_SUBTYPES = ["replaceLine", "replaceLines", "insertAfter"] as const;
|
||||
|
||||
function countHashlineEditSubtypes(args: unknown): Record<string, number> {
|
||||
const counts: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map((k) => [k, 0]));
|
||||
const counts: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
if (!args || typeof args !== "object") return counts;
|
||||
const edits = (args as { edits?: unknown[] }).edits;
|
||||
if (!Array.isArray(edits)) return counts;
|
||||
@@ -121,7 +120,7 @@ async function appendNoChangeMutationHint(
|
||||
error: string,
|
||||
args: unknown,
|
||||
cwd: string,
|
||||
originalFiles: Map<string, string>
|
||||
originalFiles: Map<string, string>,
|
||||
): Promise<string> {
|
||||
if (!error.includes("No changes made")) return error;
|
||||
const editPath = getEditPathFromArgs(args);
|
||||
@@ -187,7 +186,7 @@ function buildTimeoutRetryContext(telemetry: PromptAttemptTelemetry, retryNumber
|
||||
async function evaluateMutationIntent(
|
||||
task: EditTask,
|
||||
cwd: string,
|
||||
expectedDir: string
|
||||
expectedDir: string,
|
||||
): Promise<MutationIntentValidation | null> {
|
||||
const metadata = task.metadata;
|
||||
const file = metadata?.fileName ?? task.files[0];
|
||||
@@ -280,7 +279,7 @@ function buildGuidedHashlineEdits(actual: string, expected: string): GuidedHashl
|
||||
const firstLine = actualLines[0] ?? "";
|
||||
const firstRef = `1:${computeLineHash(1, firstLine)}`;
|
||||
edits.push({
|
||||
replaceLine: { loc: firstRef, content: pendingAdded.join("\n") + "\n" + firstLine },
|
||||
replaceLine: { loc: firstRef, content: `${pendingAdded.join("\n")}\n${firstLine}` },
|
||||
});
|
||||
} else if (insertLine <= actualLines.length) {
|
||||
const afterLine = actualLines[insertLine - 2] ?? "";
|
||||
@@ -342,7 +341,7 @@ async function buildGuidedContext(
|
||||
task: EditTask,
|
||||
cwd: string,
|
||||
expectedDir: string,
|
||||
config: BenchmarkConfig
|
||||
config: BenchmarkConfig,
|
||||
): Promise<string | null> {
|
||||
if (!config.guided) return null;
|
||||
if (config.editVariant !== "hashline") return null;
|
||||
@@ -374,7 +373,7 @@ async function buildGuidedContext(
|
||||
return [
|
||||
`Target file: \`${file}\`${metaParts.length > 0 ? ` (${metaParts.join(", ")})` : ""}.`,
|
||||
"Apply this edit tool call (single call; copy/paste args exactly):",
|
||||
"```diff\n" + argsText + "\n```",
|
||||
`\`\`\`diff\n${argsText}\n\`\`\``,
|
||||
].join("\n\n");
|
||||
}
|
||||
|
||||
@@ -510,7 +509,7 @@ async function copyFixtures(task: EditTask, destDir: string): Promise<void> {
|
||||
} else if (task.inputDir) {
|
||||
const entries = await fs.readdir(task.inputDir, { withFileTypes: true });
|
||||
await Promise.all(
|
||||
entries.map((entry) => fs.cp(join(task.inputDir!, entry.name), join(destDir, entry.name), { recursive: true }))
|
||||
entries.map(entry => fs.cp(join(task.inputDir!, entry.name), join(destDir, entry.name), { recursive: true })),
|
||||
);
|
||||
} else {
|
||||
throw new Error(`Task ${task.id} has neither tarballPath nor inputDir`);
|
||||
@@ -541,7 +540,7 @@ async function runSingleTask(
|
||||
config: BenchmarkConfig,
|
||||
cwd: string,
|
||||
expectedDir: string,
|
||||
cliPath: string
|
||||
cliPath: string,
|
||||
): Promise<TaskRunResult> {
|
||||
const startTime = Date.now();
|
||||
let client: RpcClient | null = null;
|
||||
@@ -554,10 +553,10 @@ async function runSingleTask(
|
||||
let tokens: TokenStats = { input: 0, output: 0, total: 0 };
|
||||
let agentResponse: string | undefined;
|
||||
let diff: string | undefined;
|
||||
let editFailures: EditFailure[] = [];
|
||||
const editFailures: EditFailure[] = [];
|
||||
let timeoutTelemetry: PromptAttemptTelemetry | undefined;
|
||||
let mutationIntentValidation: MutationIntentValidation | null = null;
|
||||
let toolStats = {
|
||||
const toolStats = {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
write: 0,
|
||||
@@ -565,11 +564,11 @@ async function runSingleTask(
|
||||
editFailures: 0,
|
||||
totalInputChars: 0,
|
||||
};
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map((k) => [k, 0]));
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
|
||||
const logFile = join(TMP, `run-${task.id}-${runIndex}.jsonl`);
|
||||
const logEvent = async (event: unknown) => {
|
||||
await fs.appendFile(logFile, JSON.stringify(event) + "\n");
|
||||
await fs.appendFile(logFile, `${JSON.stringify(event)}\n`);
|
||||
};
|
||||
const originalFiles = await collectOriginalFileContents(cwd, task.files);
|
||||
|
||||
@@ -584,7 +583,8 @@ async function runSingleTask(
|
||||
env.PI_EDIT_FUZZY = config.editFuzzy === "auto" ? "auto" : config.editFuzzy ? "1" : "0";
|
||||
}
|
||||
if (config.editFuzzyThreshold !== undefined) {
|
||||
env.PI_EDIT_FUZZY_THRESHOLD = config.editFuzzyThreshold === "auto" ? "auto" : String(config.editFuzzyThreshold);
|
||||
env.PI_EDIT_FUZZY_THRESHOLD =
|
||||
config.editFuzzyThreshold === "auto" ? "auto" : String(config.editFuzzyThreshold);
|
||||
}
|
||||
|
||||
client = new RpcClient({
|
||||
@@ -620,7 +620,7 @@ async function runSingleTask(
|
||||
|
||||
await fs.appendFile(
|
||||
logFile,
|
||||
`{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`
|
||||
`{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`,
|
||||
);
|
||||
|
||||
const statsBefore = await client.getSessionStats();
|
||||
@@ -634,6 +634,7 @@ async function runSingleTask(
|
||||
if (timeoutRetriesUsed < timeoutRetryLimit) {
|
||||
timeoutRetriesUsed += 1;
|
||||
retryContext = buildTimeoutRetryContext(err.telemetry, timeoutRetriesUsed, timeoutRetryLimit);
|
||||
attempt--; // Don't consume a regular attempt slot for timeout retries
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -685,7 +686,7 @@ async function runSingleTask(
|
||||
extractToolErrorMessage(e.result),
|
||||
args,
|
||||
cwd,
|
||||
originalFiles
|
||||
originalFiles,
|
||||
);
|
||||
editFailures.push({ toolCallId: e.toolCallId, args, error });
|
||||
} else {
|
||||
@@ -739,7 +740,8 @@ async function runSingleTask(
|
||||
const mustUseEditTool = Boolean(config.requireEditToolCall) && !config.noEditRequired;
|
||||
const mustUseReadTool = Boolean(config.requireReadToolCall) && !config.noEditRequired;
|
||||
const editSucceeded = toolStats.editSuccesses > 0;
|
||||
const success = verificationPassed && (!mustUseEditTool || editSucceeded) && (!mustUseReadTool || toolStats.read > 0);
|
||||
const success =
|
||||
verificationPassed && (!mustUseEditTool || editSucceeded) && (!mustUseReadTool || toolStats.read > 0);
|
||||
const metadata = task.metadata;
|
||||
|
||||
await logEvent({
|
||||
@@ -785,7 +787,7 @@ async function runBatchedTask(
|
||||
config: BenchmarkConfig,
|
||||
cwd: string,
|
||||
expectedDir: string,
|
||||
client: RpcClient
|
||||
client: RpcClient,
|
||||
): Promise<TaskRunResult> {
|
||||
const startTime = Date.now();
|
||||
const task = item.task;
|
||||
@@ -799,10 +801,10 @@ async function runBatchedTask(
|
||||
let tokens: TokenStats = { input: 0, output: 0, total: 0 };
|
||||
let agentResponse: string | undefined;
|
||||
let diff: string | undefined;
|
||||
let editFailures: EditFailure[] = [];
|
||||
const editFailures: EditFailure[] = [];
|
||||
let timeoutTelemetry: PromptAttemptTelemetry | undefined;
|
||||
let mutationIntentValidation: MutationIntentValidation | null = null;
|
||||
let toolStats = {
|
||||
const toolStats = {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
write: 0,
|
||||
@@ -810,18 +812,18 @@ async function runBatchedTask(
|
||||
editFailures: 0,
|
||||
totalInputChars: 0,
|
||||
};
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map((k) => [k, 0]));
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
|
||||
const logFile = join(TMP, `run-${task.id}-${runIndex}.jsonl`);
|
||||
const logEvent = async (event: unknown) => {
|
||||
await fs.appendFile(logFile, JSON.stringify(event) + "\n");
|
||||
await fs.appendFile(logFile, `${JSON.stringify(event)}\n`);
|
||||
};
|
||||
const originalFiles = await collectOriginalFileContents(cwd, task.files);
|
||||
|
||||
try {
|
||||
await fs.appendFile(
|
||||
logFile,
|
||||
`{"type":"meta","task":"${task.id}","run":${runIndex},"workDir":"${cwd}","batched":true}\n`
|
||||
`{"type":"meta","task":"${task.id}","run":${runIndex},"workDir":"${cwd}","batched":true}\n`,
|
||||
);
|
||||
|
||||
const maxAttempts = Math.max(1, Math.floor(config.maxAttempts ?? 1));
|
||||
@@ -840,7 +842,7 @@ async function runBatchedTask(
|
||||
});
|
||||
await fs.appendFile(
|
||||
logFile,
|
||||
`{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`
|
||||
`{"type":"prompt","attempt":${attempt + 1},"message":${JSON.stringify(promptWithContext)}}\n`,
|
||||
);
|
||||
|
||||
const statsBefore = await client.getSessionStats();
|
||||
@@ -854,6 +856,7 @@ async function runBatchedTask(
|
||||
if (timeoutRetriesUsed < timeoutRetryLimit) {
|
||||
timeoutRetriesUsed += 1;
|
||||
retryContext = buildTimeoutRetryContext(err.telemetry, timeoutRetriesUsed, timeoutRetryLimit);
|
||||
attempt--; // Don't consume a regular attempt slot for timeout retries
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -902,7 +905,7 @@ async function runBatchedTask(
|
||||
extractToolErrorMessage(e.result),
|
||||
args,
|
||||
cwd,
|
||||
originalFiles
|
||||
originalFiles,
|
||||
);
|
||||
editFailures.push({ toolCallId: e.toolCallId, args, error: toolError });
|
||||
} else {
|
||||
@@ -950,7 +953,8 @@ async function runBatchedTask(
|
||||
const mustUseEditTool = Boolean(config.requireEditToolCall) && !config.noEditRequired;
|
||||
const mustUseReadTool = Boolean(config.requireReadToolCall) && !config.noEditRequired;
|
||||
const editSucceeded = toolStats.editSuccesses > 0;
|
||||
const success = verificationPassed && (!mustUseEditTool || editSucceeded) && (!mustUseReadTool || toolStats.read > 0);
|
||||
const success =
|
||||
verificationPassed && (!mustUseEditTool || editSucceeded) && (!mustUseReadTool || toolStats.read > 0);
|
||||
const metadata = task.metadata;
|
||||
|
||||
await logEvent({
|
||||
@@ -1041,7 +1045,7 @@ function buildRunBatches(items: TaskRunItem[]): TaskRunItem[][] {
|
||||
for (let i = 0; i < pending.length && batch.length < targetSize; ) {
|
||||
const item = pending[i]!;
|
||||
const files = taskFileKeys(item.task);
|
||||
if (files.some((file) => usedFiles.has(file))) {
|
||||
if (files.some(file => usedFiles.has(file))) {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
@@ -1066,7 +1070,7 @@ async function collectPromptEvents(
|
||||
client: RpcClient,
|
||||
prompt: string,
|
||||
config: BenchmarkConfig,
|
||||
logEvent: (event: unknown) => Promise<void>
|
||||
logEvent: (event: unknown) => Promise<void>,
|
||||
): Promise<Array<{ type: string; [key: string]: unknown }>> {
|
||||
const events: Array<{ type: string; [key: string]: unknown }> = [];
|
||||
let unsubscribe: (() => void) | undefined;
|
||||
@@ -1116,11 +1120,11 @@ async function collectPromptEvents(
|
||||
lastEventType,
|
||||
recentEventTypes: [...recentEventTypes],
|
||||
pendingRetry,
|
||||
})
|
||||
}),
|
||||
);
|
||||
}, config.timeout);
|
||||
|
||||
unsubscribe = client.onEvent(async (event) => {
|
||||
unsubscribe = client.onEvent(async event => {
|
||||
if (!event) {
|
||||
return;
|
||||
}
|
||||
@@ -1177,7 +1181,7 @@ async function collectPromptEvents(
|
||||
|
||||
function diffTokenStats(
|
||||
before: { tokens: { input: number; output: number; total: number } },
|
||||
after: { tokens: { input: number; output: number; total: number } }
|
||||
after: { tokens: { input: number; output: number; total: number } },
|
||||
): TokenStats {
|
||||
const input = Math.max(0, after.tokens.input - before.tokens.input);
|
||||
const output = Math.max(0, after.tokens.output - before.tokens.output);
|
||||
@@ -1188,7 +1192,7 @@ function diffTokenStats(
|
||||
function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult {
|
||||
const orderedRuns = runs.slice().sort((a, b) => a.runIndex - b.runIndex);
|
||||
const n = orderedRuns.length;
|
||||
const successfulRuns = orderedRuns.filter((r) => r.success).length;
|
||||
const successfulRuns = orderedRuns.filter(r => r.success).length;
|
||||
const successRate = n > 0 ? successfulRuns / n : 0;
|
||||
|
||||
const avgTokens: TokenStats =
|
||||
@@ -1202,7 +1206,7 @@ function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult {
|
||||
|
||||
const avgDuration = n > 0 ? Math.round(orderedRuns.reduce((sum, r) => sum + r.duration, 0) / n) : 0;
|
||||
const indentScores = orderedRuns
|
||||
.map((run) => run.indentScore)
|
||||
.map(run => run.indentScore)
|
||||
.filter((score): score is number => typeof score === "number");
|
||||
const avgIndentScore =
|
||||
indentScores.length > 0 ? indentScores.reduce((sum, score) => sum + score, 0) / indentScores.length : 0;
|
||||
@@ -1262,7 +1266,7 @@ async function runBatch(
|
||||
items: TaskRunItem[],
|
||||
config: BenchmarkConfig,
|
||||
cliPath: string,
|
||||
onProgress?: (event: ProgressEvent) => void
|
||||
onProgress?: (event: ProgressEvent) => void,
|
||||
): Promise<Array<{ task: EditTask; result: TaskRunResult }>> {
|
||||
const workDir = join(TMP, `batch-${crypto.randomUUID()}`);
|
||||
await fs.mkdir(workDir, { recursive: true });
|
||||
@@ -1275,13 +1279,13 @@ async function runBatch(
|
||||
|
||||
try {
|
||||
await Promise.all(
|
||||
orderedItems.map(async (item) => {
|
||||
orderedItems.map(async item => {
|
||||
const expected = await getExpectedDir(item.task);
|
||||
expectedDirs.set(item.task.id, expected);
|
||||
})
|
||||
}),
|
||||
);
|
||||
|
||||
await Promise.all(orderedItems.map((item) => copyFixtures(item.task, workDir)));
|
||||
await Promise.all(orderedItems.map(item => copyFixtures(item.task, workDir)));
|
||||
|
||||
const env: Record<string, string> = { PI_NO_TITLE: "1" };
|
||||
if (config.editVariant !== undefined) {
|
||||
@@ -1291,7 +1295,8 @@ async function runBatch(
|
||||
env.PI_EDIT_FUZZY = config.editFuzzy === "auto" ? "auto" : config.editFuzzy ? "1" : "0";
|
||||
}
|
||||
if (config.editFuzzyThreshold !== undefined) {
|
||||
env.PI_EDIT_FUZZY_THRESHOLD = config.editFuzzyThreshold === "auto" ? "auto" : String(config.editFuzzyThreshold);
|
||||
env.PI_EDIT_FUZZY_THRESHOLD =
|
||||
config.editFuzzyThreshold === "auto" ? "auto" : String(config.editFuzzyThreshold);
|
||||
}
|
||||
|
||||
client = new RpcClient({
|
||||
@@ -1352,7 +1357,7 @@ async function runBatch(
|
||||
export async function runTask(
|
||||
task: EditTask,
|
||||
config: BenchmarkConfig,
|
||||
onProgress?: (event: ProgressEvent) => void
|
||||
onProgress?: (event: ProgressEvent) => void,
|
||||
): Promise<TaskResult> {
|
||||
const tempDirs: TempDir[] = [];
|
||||
const { dir: expectedDir, cleanup: cleanupExpected } = await getExpectedDir(task);
|
||||
@@ -1390,11 +1395,11 @@ export async function runTask(
|
||||
export async function runBenchmark(
|
||||
tasks: EditTask[],
|
||||
config: BenchmarkConfig,
|
||||
onProgress?: (event: ProgressEvent) => void
|
||||
onProgress?: (event: ProgressEvent) => void,
|
||||
): Promise<BenchmarkResult> {
|
||||
const startTime = new Date().toISOString();
|
||||
const runItems: TaskRunItem[] = tasks.flatMap((task) =>
|
||||
Array.from({ length: config.runsPerTask }, (_, runIndex) => ({ task, runIndex }))
|
||||
const runItems: TaskRunItem[] = tasks.flatMap(task =>
|
||||
Array.from({ length: config.runsPerTask }, (_, runIndex) => ({ task, runIndex })),
|
||||
);
|
||||
|
||||
const batches = buildRunBatches(runItems);
|
||||
@@ -1423,13 +1428,13 @@ export async function runBenchmark(
|
||||
|
||||
await Promise.all(running);
|
||||
|
||||
const taskResults = tasks.map((task) => summarizeTaskRuns(task, resultsByTask.get(task.id) ?? []));
|
||||
const taskResults = tasks.map(task => summarizeTaskRuns(task, resultsByTask.get(task.id) ?? []));
|
||||
|
||||
const endTime = new Date().toISOString();
|
||||
|
||||
const allRuns = taskResults.flatMap((t) => t.runs);
|
||||
const allRuns = taskResults.flatMap(t => t.runs);
|
||||
const totalRuns = allRuns.length;
|
||||
const successfulRuns = allRuns.filter((r) => r.success).length;
|
||||
const successfulRuns = allRuns.filter(r => r.success).length;
|
||||
|
||||
const totalTokens: TokenStats = {
|
||||
input: allRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
@@ -1439,7 +1444,7 @@ export async function runBenchmark(
|
||||
|
||||
const totalDuration = allRuns.reduce((sum, r) => sum + r.duration, 0);
|
||||
const indentScores = allRuns
|
||||
.map((run) => run.indentScore)
|
||||
.map(run => run.indentScore)
|
||||
.filter((score): score is number => typeof score === "number");
|
||||
const avgIndentScore =
|
||||
indentScores.length > 0 ? indentScores.reduce((sum, score) => sum + score, 0) / indentScores.length : 0;
|
||||
@@ -1454,20 +1459,20 @@ export async function runBenchmark(
|
||||
};
|
||||
|
||||
const editSuccessRate = totalToolCalls.edit > 0 ? totalToolCalls.editSuccesses / totalToolCalls.edit : 1;
|
||||
const timeoutRuns = allRuns.filter((r) => r.error?.includes("Timeout waiting for agent_end")).length;
|
||||
const runsWithMutationIntent = allRuns.filter((r) => typeof r.mutationIntentMatched === "boolean");
|
||||
const timeoutRuns = allRuns.filter(r => r.error?.includes("Timeout waiting for agent_end")).length;
|
||||
const runsWithMutationIntent = allRuns.filter(r => typeof r.mutationIntentMatched === "boolean");
|
||||
const mutationIntentMatchRate =
|
||||
runsWithMutationIntent.length > 0
|
||||
? runsWithMutationIntent.filter((r) => r.mutationIntentMatched).length / runsWithMutationIntent.length
|
||||
? runsWithMutationIntent.filter(r => r.mutationIntentMatched).length / runsWithMutationIntent.length
|
||||
: undefined;
|
||||
|
||||
const hashlineEditSubtypes: Record<string, number> | undefined =
|
||||
config.editVariant === "hashline"
|
||||
? Object.fromEntries(
|
||||
HASHLINE_SUBTYPES.map((key) => [
|
||||
HASHLINE_SUBTYPES.map(key => [
|
||||
key,
|
||||
allRuns.reduce((sum, r) => sum + (r.hashlineEditSubtypes?.[key] ?? 0), 0),
|
||||
])
|
||||
]),
|
||||
)
|
||||
: undefined;
|
||||
|
||||
@@ -1476,8 +1481,8 @@ export async function runBenchmark(
|
||||
totalRuns,
|
||||
successfulRuns,
|
||||
overallSuccessRate: successfulRuns / totalRuns,
|
||||
tasksWithAllPassing: taskResults.filter((t) => t.successRate === 1).length,
|
||||
tasksWithAnyFailing: taskResults.filter((t) => t.successRate < 1).length,
|
||||
tasksWithAllPassing: taskResults.filter(t => t.successRate === 1).length,
|
||||
tasksWithAnyFailing: taskResults.filter(t => t.successRate < 1).length,
|
||||
totalTokens,
|
||||
avgTokensPerRun: {
|
||||
input: Math.round(totalTokens.input / totalRuns),
|
||||
|
||||
@@ -1,9 +1,4 @@
|
||||
/**
|
||||
* Tarball utilities for reading fixtures directly from .tar.gz archives.
|
||||
* Uses Bun.Archive for native tar.gz handling.
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import { basename, dirname, join } from "node:path";
|
||||
import { basename, join } from "node:path";
|
||||
|
||||
export interface TarballTask {
|
||||
id: string;
|
||||
@@ -154,11 +149,11 @@ export async function loadTasksFromTarball(tarballPath: string): Promise<Tarball
|
||||
const entries = await readTarball(tarballPath);
|
||||
const { tasks, issues } = parseTarballEntries(entries);
|
||||
if (issues.length > 0) {
|
||||
const details = issues.map((issue) => `- ${issue.taskId}: ${issue.message}`).join("\n");
|
||||
const details = issues.map(issue => `- ${issue.taskId}: ${issue.message}`).join("\n");
|
||||
throw new Error(`Fixture tarball validation failed:\n${details}`);
|
||||
}
|
||||
|
||||
const normalized: TarballTask[] = tasks.map((task) => ({
|
||||
const normalized: TarballTask[] = tasks.map(task => ({
|
||||
id: task.id,
|
||||
prompt: task.prompt?.trim() ?? "",
|
||||
metadata: task.metadata ?? {},
|
||||
|
||||
@@ -9,10 +9,10 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import { basename, join } from "node:path";
|
||||
import {
|
||||
type FixtureValidationIssue,
|
||||
type TarballTask,
|
||||
extractTaskFiles,
|
||||
type FixtureValidationIssue,
|
||||
loadTasksFromTarball,
|
||||
type TarballTask,
|
||||
validateTarballFixtures,
|
||||
} from "./tarball";
|
||||
|
||||
@@ -48,7 +48,7 @@ export const DEFAULT_TARBALL_PATH = join(import.meta.dir, "fixtures.tar.gz");
|
||||
function titleize(id: string): string {
|
||||
return id
|
||||
.split(/[-_]/)
|
||||
.map((part) => (part ? part[0].toUpperCase() + part.slice(1) : part))
|
||||
.map(part => (part ? part[0].toUpperCase() + part.slice(1) : part))
|
||||
.join(" ");
|
||||
}
|
||||
|
||||
@@ -132,7 +132,7 @@ function tarballTaskToEditTask(task: TarballTask, tarballPath: string): EditTask
|
||||
|
||||
export async function loadTasks(): Promise<EditTask[]> {
|
||||
const tarballTasks = await loadTasksFromTarball(DEFAULT_TARBALL_PATH);
|
||||
return tarballTasks.map((t) => tarballTaskToEditTask(t, DEFAULT_TARBALL_PATH));
|
||||
return tarballTasks.map(t => tarballTaskToEditTask(t, DEFAULT_TARBALL_PATH));
|
||||
}
|
||||
|
||||
export { extractTaskFiles };
|
||||
@@ -218,13 +218,13 @@ export async function validateFixturesFromDir(fixturesPath: string): Promise<Fix
|
||||
continue;
|
||||
}
|
||||
const fileName = basename(metadata.file_path);
|
||||
if (!inputFiles.some((file) => basename(file) === fileName)) {
|
||||
if (!inputFiles.some(file => basename(file) === fileName)) {
|
||||
issues.push({
|
||||
taskId,
|
||||
message: `metadata file_path ${metadata.file_path} not found in input files`,
|
||||
});
|
||||
}
|
||||
if (!expectedFiles.some((file) => basename(file) === fileName)) {
|
||||
if (!expectedFiles.some(file => basename(file) === fileName)) {
|
||||
issues.push({
|
||||
taskId,
|
||||
message: `metadata file_path ${metadata.file_path} not found in expected files`,
|
||||
|
||||
@@ -109,9 +109,9 @@ export async function verifyExpectedFileSubset(
|
||||
const expectedFiles = files?.length ? files.slice().sort() : expectedFixtureFiles;
|
||||
const actualFiles = await listFiles(actualDir);
|
||||
|
||||
const missingFiles = expectedFiles.filter((file) => !actualFiles.includes(file));
|
||||
const extraFiles = actualFiles.filter((file) => !expectedFiles.includes(file));
|
||||
const missingExpected = expectedFiles.filter((file) => !expectedFixtureFiles.includes(file));
|
||||
const missingFiles = expectedFiles.filter(file => !actualFiles.includes(file));
|
||||
const extraFiles = actualFiles.filter(file => !expectedFiles.includes(file));
|
||||
const missingExpected = expectedFiles.filter(file => !expectedFixtureFiles.includes(file));
|
||||
|
||||
if (missingExpected.length > 0) {
|
||||
return {
|
||||
|
||||
Reference in New Issue
Block a user