212d56bc11
- Added `toolStrictMode` support with `all_strict`/`none`/`mixed` options to OpenAI compatibility. - Fixed OpenAI-completion strict-mode flows by capturing failed HTTP responses and retrying once as non-strict. - Fixed completion error reporting by surfacing captured status, headers, and JSON `type`/`param`/`code` details. - Improved strict-schema enforcement with WeakMap memoization and circular-schema detection in sanitization. - Fixed OpenRouter provider lookup by resolving fallback model IDs for suffix and date variants in registry resolution. - Refactored benchmark tooling and added async RPC error-window tracking for scheduled run execution.
37 lines
1.0 KiB
Python
Executable File
37 lines
1.0 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Chunk edit benchmark: tests chunk-mode edit tool usage across models with a simple edit task.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from edit_benchmark_common import BenchmarkSpec, EDIT_DIFF, EXPECTED_CONTENT, run_benchmark_main
|
|
|
|
EDIT_PROMPT = f"""\
|
|
Use the `read` tool to inspect `test.py`, then use the `edit` tool in chunk mode to make `test.py` exactly match the requested change.
|
|
|
|
Apply this diff:
|
|
```diff
|
|
{EDIT_DIFF}```
|
|
|
|
Final expected file content:
|
|
```python
|
|
{EXPECTED_CONTENT}```
|
|
"""
|
|
|
|
CHUNK_BENCHMARK = BenchmarkSpec(
|
|
description="Benchmark chunk-mode edit tool across models with simple edit tasks.",
|
|
workspace_prefix="chunk-benchmark",
|
|
tools=("edit", "read"),
|
|
env={"PI_EDIT_VARIANT": "chunk", "PI_STRICT_EDIT_MODE": "1"},
|
|
initial_prompt=EDIT_PROMPT,
|
|
retry_instruction='Use `read(path="test.py")` to refresh chunk selectors if needed, then try again using the edit tool.',
|
|
)
|
|
|
|
|
|
def main() -> int:
|
|
return run_benchmark_main(CHUNK_BENCHMARK)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|