feat(scripts): added unified edit benchmark runner with variant selection

- Replaced the standalone chunk and vim benchmark scripts with a unified `scripts/edit-benchmark.py` entrypoint.
- Added `--variant` and `PI_EDIT_VARIANT` handling with validation for supported edit modes.
- Generated benchmark specs dynamically per variant with mode-specific prompts and retry instructions.
This commit is contained in:
can1357
2026-04-24 20:00:20 +02:00
parent 83e897d5ea
commit 326169f0f0
3 changed files with 90 additions and 73 deletions
-36
View File
@@ -1,36 +0,0 @@
#!/usr/bin/env python3
"""
Chunk edit benchmark: tests chunk-mode edit tool usage across models with a simple edit task.
"""
from __future__ import annotations
from edit_benchmark_common import BenchmarkSpec, EDIT_DIFF, EXPECTED_CONTENT, run_benchmark_main
EDIT_PROMPT = f"""\
Use the `read` tool to inspect `test.rs`, then use the `edit` tool in chunk mode to make `test.rs` exactly match the requested change.
Apply this diff:
```diff
{EDIT_DIFF}```
Final expected file content:
```python
{EXPECTED_CONTENT}```
"""
CHUNK_BENCHMARK = BenchmarkSpec(
description="Benchmark chunk-mode edit tool across models with simple edit tasks.",
workspace_prefix="chunk-benchmark",
tools=("edit", "read"),
env={"PI_EDIT_VARIANT": "chunk", "PI_STRICT_EDIT_MODE": "1"},
initial_prompt=EDIT_PROMPT,
retry_instruction='Use `read(path="test.rs")` to refresh chunk selectors if needed, then try again using the edit tool.',
)
def main() -> int:
return run_benchmark_main(CHUNK_BENCHMARK)
if __name__ == "__main__":
raise SystemExit(main())
+90
View File
@@ -0,0 +1,90 @@
#!/usr/bin/env python3
"""
Edit benchmark: tests the edit tool across models with a simple edit task.
Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `chunk`, `vim`,
`hashline`, `replace`, `patch`, `apply_patch`) or `--variant`.
Examples:
PI_EDIT_VARIANT=chunk scripts/edit-benchmark.py
PI_EDIT_VARIANT=vim scripts/edit-benchmark.py
scripts/edit-benchmark.py --variant hashline
"""
from __future__ import annotations
import os
import sys
from edit_benchmark_common import BenchmarkSpec, EDIT_DIFF, EXPECTED_CONTENT, run_benchmark_main
# Variants recognised by packages/coding-agent/src/utils/edit-mode.ts.
VALID_VARIANTS = ("replace", "patch", "hashline", "chunk", "vim", "apply_patch")
def _extract_variant_arg() -> str | None:
"""Pop `--variant <value>` (or `--variant=<value>`) from sys.argv before argparse in common runs."""
argv = sys.argv
for i, arg in enumerate(argv[1:], start=1):
if arg == "--variant" and i + 1 < len(argv):
value = argv[i + 1]
del argv[i : i + 2]
return value
if arg.startswith("--variant="):
value = arg.split("=", 1)[1]
del argv[i]
return value
return None
def _resolve_variant() -> str:
cli_variant = _extract_variant_arg()
variant = cli_variant or os.environ.get("PI_EDIT_VARIANT")
if not variant:
raise SystemExit(
"edit-benchmark: set PI_EDIT_VARIANT=<variant> or pass --variant <variant>.\n"
f"Valid variants: {', '.join(VALID_VARIANTS)}"
)
if variant not in VALID_VARIANTS:
raise SystemExit(
f"edit-benchmark: unknown variant '{variant}'.\n"
f"Valid variants: {', '.join(VALID_VARIANTS)}"
)
return variant
def build_spec(variant: str) -> BenchmarkSpec:
mode_phrase = f"in {variant} mode"
prompt = (
f"Use the `read` tool to inspect `test.rs`, then use the `edit` tool {mode_phrase} "
f"to make `test.rs` exactly match the requested change.\n"
f"\n"
f"Apply this diff:\n"
f"```diff\n"
f"{EDIT_DIFF}```\n"
f"\n"
f"Final expected file content:\n"
f"```rust\n"
f"{EXPECTED_CONTENT}```\n"
)
retry = (
'Use `read(path="test.rs")` to refresh chunk selectors if needed, then try again using the edit tool.'
if variant == "chunk"
else f"Please try again using the edit tool {mode_phrase}."
)
return BenchmarkSpec(
description=f"Benchmark edit tool in {variant} mode across models with simple edit tasks.",
workspace_prefix=f"{variant}-benchmark",
tools=("edit", "read"),
env={"PI_EDIT_VARIANT": variant, "PI_STRICT_EDIT_MODE": "1"},
initial_prompt=prompt,
retry_instruction=retry,
)
def main() -> int:
variant = _resolve_variant()
return run_benchmark_main(build_spec(variant))
if __name__ == "__main__":
raise SystemExit(main())
-37
View File
@@ -1,37 +0,0 @@
#!/usr/bin/env python3
"""
Vim edit-mode benchmark: tests the edit tool in vim mode across models with a simple edit task.
"""
from __future__ import annotations
from edit_benchmark_common import BenchmarkSpec, EDIT_DIFF, EXPECTED_CONTENT, run_benchmark_main
EDIT_PROMPT = f"""\
Use the `read` tool to inspect `test.rs`, then use the `edit` tool in vim mode to make `test.rs` exactly match the requested change.
Apply this diff:
```diff
{EDIT_DIFF}```
Final expected file content:
```rust
{EXPECTED_CONTENT}```
"""
VIM_BENCHMARK = BenchmarkSpec(
description="Benchmark edit tool in vim mode across models with simple edit tasks.",
workspace_prefix="vim-benchmark",
tools=("edit", "read"),
env={"PI_EDIT_VARIANT": "vim", "PI_STRICT_EDIT_MODE": "1"},
initial_prompt=EDIT_PROMPT,
retry_instruction="Please try again using the edit tool in vim mode.",
)
def main() -> int:
return run_benchmark_main(VIM_BENCHMARK)
if __name__ == "__main__":
raise SystemExit(main())