212d56bc11
- Added `toolStrictMode` support with `all_strict`/`none`/`mixed` options to OpenAI compatibility. - Fixed OpenAI-completion strict-mode flows by capturing failed HTTP responses and retrying once as non-strict. - Fixed completion error reporting by surfacing captured status, headers, and JSON `type`/`param`/`code` details. - Improved strict-schema enforcement with WeakMap memoization and circular-schema detection in sanitization. - Fixed OpenRouter provider lookup by resolving fallback model IDs for suffix and date variants in registry resolution. - Refactored benchmark tooling and added async RPC error-window tracking for scheduled run execution.
31 lines
832 B
Python
Executable File
31 lines
832 B
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Vim edit benchmark: tests the vim tool across models with a simple edit task.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from edit_benchmark_common import BenchmarkSpec, EDIT_DIFF, run_benchmark_main
|
|
|
|
EDIT_PROMPT = f"""\
|
|
Apply the following diff to the file `test.py` using the vim tool with the minimum amount of "moves":
|
|
```diff
|
|
{EDIT_DIFF}```
|
|
"""
|
|
|
|
VIM_BENCHMARK = BenchmarkSpec(
|
|
description="Benchmark vim tool across models with simple edit tasks.",
|
|
workspace_prefix="vim-benchmark",
|
|
tools=("edit", "read"),
|
|
env={"PI_EDIT_VARIANT": "vim", "PI_STRICT_EDIT_MODE": "1"},
|
|
initial_prompt=EDIT_PROMPT,
|
|
retry_instruction="Please try again using the vim tool.",
|
|
)
|
|
|
|
|
|
def main() -> int:
|
|
return run_benchmark_main(VIM_BENCHMARK)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|