diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 8fa086c08..c93adedf6 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -127,6 +127,9 @@ jobs:
- uses: Swatinem/rust-cache@v2
with:
cache-workspace-crates: true
+ - uses: taiki-e/install-action@v2
+ with:
+ tool: nextest
- uses: mlugg/setup-zig@v2
with:
version: 0.15.2
diff --git a/bun.lock b/bun.lock
index 7edba654f..a58c07137 100644
--- a/bun.lock
+++ b/bun.lock
@@ -8,10 +8,11 @@
"@biomejs/biome": "^2.4",
"@bufbuild/protoc-gen-es": "^2.11",
"@types/bun": "^1.3",
- "@typescript/native-preview": "^7.0.0-dev.20260302.1",
+ "@typescript/native-preview": "7.0.0-dev.20260322.1",
"lint-staged": "^16.3",
"prettier": "^3.8",
"turbo": "^2.5.8",
+ "typescript": "^6.0.2",
},
},
"packages/agent": {
@@ -662,21 +663,21 @@
"@types/yauzl": ["@types/yauzl@2.10.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-oJoftv0LSuaDZE3Le4DbKX+KS9G36NzOeSap90UIK0yMA/NhKJhqlSGtNDORNRaIbQfzjXDrQa0ytJ6mNRGz/Q=="],
- "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260410.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260410.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260410.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260410.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260410.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260410.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260410.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260410.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-K3TIwBw4XGQM33wW8KUqRU7r6ZY1IqB8chk1u1kT+CDj4iu+eQ6jCXgU7EDxmpJ++gbNcIf8iBYgWgYNssrhZQ=="],
+ "@typescript/native-preview": ["@typescript/native-preview@7.0.0-dev.20260322.1", "", { "optionalDependencies": { "@typescript/native-preview-darwin-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-darwin-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-linux-x64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-arm64": "7.0.0-dev.20260322.1", "@typescript/native-preview-win32-x64": "7.0.0-dev.20260322.1" }, "bin": { "tsgo": "bin/tsgo.js" } }, "sha512-CmzQTKvesYHmz3g92G+XPDis25ocvHqa/gK8m98w+bML99KJLEWQKVlvkLrYA85JiJEK+XBIiz+6lCgUqRkWXA=="],
- "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260410.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-bpLYm6woXd8BECzV9AQvPqISVeohpekK1qwpRopvNIxydhRQ4fEjZsS7EtDYpqHAW4/u1uEv07P9/iS6TAL1fQ=="],
+ "@typescript/native-preview-darwin-arm64": ["@typescript/native-preview-darwin-arm64@7.0.0-dev.20260322.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-5wSilxwLGX5fMKJgsUkCBwOfW9GMG3WF5j77CVBOdFI7miFaR3JQaPzTA+uyHDMNIIeSDo1KtV77GT48Y/d0Xg=="],
- "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260410.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-V8bW8g5hgu+bAwGvTqF1kilkkoDgxhxi5egrdMUeWQkR+MIisoBQeaAupqMpLoSkqZsc/kKucM0zwBNC/KRU3Q=="],
+ "@typescript/native-preview-darwin-x64": ["@typescript/native-preview-darwin-x64@7.0.0-dev.20260322.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-G806SrfxkYNAgZ9Xk53+OvbmIg9iD5hjaiD2QhDQL2aZjzy10D4MhcdaZEOoMfw0OI/PoJPYOiPD+9/x2kw3Lg=="],
- "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260410.1", "", { "os": "linux", "cpu": "arm" }, "sha512-NO6Ci65ADadOCr2ycTxOyCgC5kyk+Ryjl8k5c78mz9sKDxYqwEtryFFjLqitAG+rejtJbnUq897WRICjAOwslA=="],
+ "@typescript/native-preview-linux-arm": ["@typescript/native-preview-linux-arm@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "arm" }, "sha512-0a12pp19ELiNHMqTglfQQQNMsxvtzpjAa4qf12oMJoGyy+UnguKEmaaaCHdp75KvBXGDzlssfDAdiy+NirN19A=="],
- "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260410.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-MOluRRAhv46s9ScFmePa0InMHmpZ/z0Evc11RrTKsg+bN8BR7sWoAtFq6IujEDK9WVP7YmEYtBRgEfMLuqVojw=="],
+ "@typescript/native-preview-linux-arm64": ["@typescript/native-preview-linux-arm64@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-+FyomEEt3K8TBO//n3Ijr61SDM2F7cxZCVqGt+Wk3rLcOCQ2i+8+p64gdsZCmImy3CyP0hBnxPydEbyNkZLtvg=="],
- "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260410.1", "", { "os": "linux", "cpu": "x64" }, "sha512-IofIUrMGjXmZKDEMaRgshzOne0EQZtx9vE/6URHfgmDnWLDKWzz9eQ2qWmvsFD2vOBbgc6GwVWEq6XTHMEfx2A=="],
+ "@typescript/native-preview-linux-x64": ["@typescript/native-preview-linux-x64@7.0.0-dev.20260322.1", "", { "os": "linux", "cpu": "x64" }, "sha512-MviQe5x4WqQGv/Vhu4hcv2A0qTW/BTaZPbOLYCtvhuovNFO6D++ZmJAbHvA0h/bJEaNTgxKZdZPHMpCfSEKfjA=="],
- "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260410.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-TXmE+wovQqRo+qAhaewB0MPB9esgayvSHr6vFlCpHykHqbDl3FUucuC4F8yU6zVOA3UqXTk4/GHeLsAvU7YEgQ=="],
+ "@typescript/native-preview-win32-arm64": ["@typescript/native-preview-win32-arm64@7.0.0-dev.20260322.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-ibnMaXDJPSgMXKC61NHiFlww/xjAEINgc1mcn2ntTfuGHwduU4P9Bi038TxXg95Wmu3v6xIPIorXXsBOdE+p3Q=="],
- "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260410.1", "", { "os": "win32", "cpu": "x64" }, "sha512-dMFT4tdHBe2vVA2WPQMjorT+fzCURRtillevQzz8/bwCEz2uXSnpu4oLRLS5045ppGE0wCFELE+Hq5z2oRddDw=="],
+ "@typescript/native-preview-win32-x64": ["@typescript/native-preview-win32-x64@7.0.0-dev.20260322.1", "", { "os": "win32", "cpu": "x64" }, "sha512-O+r1RToWBbGkK7NXC7DpraLObSWyxvSqRiSfr/BlZ351Cdq1q3121zCGzVtqERGeRtVoEMRrzS5ITOd6On/pCw=="],
"@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, "sha512-PJFXFS4ZJKiJ9Qiuix6Dz/OwEIqHD7Dme1UwZhTK11vR+5dqW2ACbdndWQexBzCx+CPuMe5WBYQWCsFyGlQLlQ=="],
@@ -1208,7 +1209,7 @@
"typed-query-selector": ["typed-query-selector@2.12.1", "", {}, "sha512-uzR+FzI8qrUEIu96oaeBJmd9E7CFEiQ3goA5qCVgc4s5llSubcfGHq9yUstZx/k4s9dXHVKsE35YWoFyvEqEHA=="],
- "typescript": ["typescript@5.4.5", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-vcI4UpRgg81oIRUFwR0WSIHKt11nJ7SAVlYNIu+QpqeyXP+gpQJy/Z4+F0aGxSE4MqwjyXvW/TzgkLAx2AGHwQ=="],
+ "typescript": ["typescript@6.0.2", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-bGdAIrZ0wiGDo5l8c++HWtbaNCWTS4UTv7RaTH/ThVIgjkveJt83m74bBHMJkuCbslY8ixgLBVZJIOiQlQTjfQ=="],
"uglify-js": ["uglify-js@3.19.3", "", { "bin": { "uglifyjs": "bin/uglifyjs" } }, "sha512-v3Xu+yuwBXisp6QYTcH4UbH+xYJXqnq2m/LtQVWKWzYc1iehYnLixoQDN9FH6/j9/oybfd6W9Ghwkl8+UMKTKQ=="],
@@ -1268,6 +1269,8 @@
"@aws-sdk/xml-builder/fast-xml-parser": ["fast-xml-parser@5.5.8", "", { "dependencies": { "fast-xml-builder": "^1.1.4", "path-expression-matcher": "^1.2.0", "strnum": "^2.2.0" }, "bin": { "fxparser": "src/cli/cli.js" } }, "sha512-Z7Fh2nVQSb2d+poDViM063ix2ZGt9jmY1nWhPfHBOK2Hgnb/OW3P4Et3P/81SEej0J7QbWtJqxO05h8QYfK7LQ=="],
+ "@bufbuild/protoplugin/typescript": ["typescript@5.4.5", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-vcI4UpRgg81oIRUFwR0WSIHKt11nJ7SAVlYNIu+QpqeyXP+gpQJy/Z4+F0aGxSE4MqwjyXvW/TzgkLAx2AGHwQ=="],
+
"chromium-bidi/zod": ["zod@3.25.76", "", {}, "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ=="],
"cliui/string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="],
diff --git a/crates/pi-natives/src/chunk/edit.rs b/crates/pi-natives/src/chunk/edit.rs
index 15185036d..b0a29b646 100644
--- a/crates/pi-natives/src/chunk/edit.rs
+++ b/crates/pi-natives/src/chunk/edit.rs
@@ -473,8 +473,10 @@ fn apply_replace(
// For prologue/epilogue replacements, ensure the replacement preserves
// the newline boundary so the body content isn't joined onto the same
// line as the replacement.
- if matches!(target.region, Some(ChunkRegion::Head | ChunkRegion::Tail | ChunkRegion::Decl))
- && !replacement.is_empty()
+ if matches!(
+ target.region,
+ Some(ChunkRegion::Head | ChunkRegion::Body | ChunkRegion::Tail | ChunkRegion::Decl)
+ ) && !replacement.is_empty()
&& !replacement.ends_with('\n')
&& state.source.as_bytes().get(region_end.saturating_sub(1)) == Some(&b'\n')
{
@@ -1269,15 +1271,21 @@ fn compute_insert_indent(
return indent_char.repeat(first_child.indent as usize);
}
- for line in chunk_slice(&state.source, anchor).split('\n').skip(1) {
- if line.trim().is_empty() {
- continue;
+ // Scan only the @body region (between prologue and epilogue), not the full
+ // chunk. This avoids picking up the closing delimiter's indent for
+ // empty/sparse bodies.
+ let (body_start, body_end) = chunk_region_range(anchor, ChunkRegion::Body);
+ if body_start < body_end && body_end <= state.source.len() {
+ for line in state.source[body_start..body_end].split('\n') {
+ if line.trim().is_empty() {
+ continue;
+ }
+ let prefix_len = line.len() - line.trim_start_matches([' ', '\t']).len();
+ if prefix_len > 0 {
+ return line[..prefix_len].to_owned();
+ }
+ break;
}
- let prefix_len = line.len() - line.trim_start_matches([' ', '\t']).len();
- if prefix_len > 0 {
- return line[..prefix_len].to_owned();
- }
- break;
}
let indent_char = if anchor.indent_char.is_empty() {
@@ -3955,4 +3963,115 @@ function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>
result.diff_after
);
}
+
+ #[test]
+ fn markdown_append_chunk_preserves_trailing_blank_line() {
+ // Two sibling sections separated by a blank line. Appending to the
+ // paragraph chunk (leaf) inside the first section must preserve the
+ // blank-line gap before the next heading.
+ let source = "# Title\n\nSome text.\n\n## Next Section\n\nMore text.\n";
+ let state = state_for(source, "markdown");
+
+ // The paragraph "Some text." is sect_Title.chunk_2.
+ let para_chunk = state
+ .inner()
+ .chunk("sect_Title.chunk_2")
+ .expect("paragraph chunk should exist");
+
+ let result = apply_single_edit(&state, "test.md", EditOperation {
+ op: ChunkEditOp::Append,
+ sel: Some(para_chunk.path.clone()),
+ crc: None,
+ region: None,
+ content: Some("Appended line.\n".to_owned()),
+ find: None,
+ });
+
+ // The blank line before ## Next Section should be preserved
+ assert!(
+ result
+ .diff_after
+ .contains("Appended line.\n\n## Next Section"),
+ "blank line before next section must be preserved after append: {:?}",
+ result.diff_after
+ );
+ }
+
+ #[test]
+ fn markdown_after_chunk_preserves_blank_line_separator() {
+ // 'after' on a table chunk followed by a blank-line separator and a
+ // heading. The blank line must survive the insertion.
+ let source = "# Section\n\n| A | B |\n|---|---|\n| 1 | 2 |\n\n## Next\n";
+ let state = state_for(source, "markdown");
+
+ // The table is sect_Sectio.chunk_2 (L3-L5).
+ let table_chunk = state
+ .inner()
+ .chunk("sect_Sectio.chunk_2")
+ .expect("table chunk should exist");
+
+ let result = apply_single_edit(&state, "test.md", EditOperation {
+ op: ChunkEditOp::After,
+ sel: Some(table_chunk.path.clone()),
+ crc: None,
+ region: None,
+ content: Some("Extra paragraph.\n".to_owned()),
+ find: None,
+ });
+
+ // Blank line before ## Next must be preserved
+ assert!(
+ result.diff_after.contains("Extra paragraph.\n\n## Next"),
+ "blank line before next heading must be preserved after 'after' insert: {:?}",
+ result.diff_after
+ );
+ }
+
+ #[test]
+ fn body_replace_nested_fn_uses_correct_indent() {
+ let source = "impl Server {\n fn is_running(&self) -> bool {\n true\n }\n}\n";
+ let state = state_for(source, "rust");
+ let chunk = state
+ .inner()
+ .tree
+ .chunks
+ .iter()
+ .find(|c| c.identifier.as_deref() == Some("is_running") || c.path.contains("is_run"))
+ .expect("is_running chunk");
+ let result = apply_single_edit(&state, "test.rs", EditOperation {
+ op: ChunkEditOp::Replace,
+ sel: Some(chunk.path.clone()),
+ crc: Some(chunk.checksum.clone()),
+ region: Some(ChunkRegion::Body),
+ content: Some("false\n".to_owned()),
+ find: None,
+ });
+ // Body should be at 2 levels of indent (8 spaces), not 1 level (4 spaces)
+ let new_source = &result.diff_after;
+ assert!(
+ new_source.contains(" false"),
+ "expected body at 8-space indent (2 levels), got:\n{new_source}"
+ );
+ }
+
+ #[test]
+ fn body_replace_preserves_closing_delimiter_on_own_line() {
+ let source = "fn foo() {\n old_body();\n}\n";
+ let state = state_for(source, "rust");
+ let chunk = state.inner().chunk("fn_foo").expect("fn_foo");
+ let result = apply_single_edit(&state, "test.rs", EditOperation {
+ op: ChunkEditOp::Replace,
+ sel: Some("fn_foo".to_owned()),
+ crc: Some(chunk.checksum.clone()),
+ region: Some(ChunkRegion::Body),
+ content: Some("new_body();".to_owned()), // No trailing newline
+ find: None,
+ });
+ let new_source = &result.diff_after;
+ // Closing } should be on its own line, not merged
+ assert!(
+ new_source.contains("new_body();\n}"),
+ "expected closing brace on own line, got:\n{new_source}"
+ );
+ }
}
diff --git a/package.json b/package.json
index 79e20e42b..8032f8cfa 100644
--- a/package.json
+++ b/package.json
@@ -15,9 +15,10 @@
"build:ws": "turbo run build",
"build:native": "bun --cwd=packages/natives run build",
"build:affected": "turbo run build --affected",
- "test": "bun run test:ts",
- "test:ts": "turbo run test",
- "test:affected": "turbo run test --affected",
+ "test": "bun run --parallel test:ts test:rs",
+ "test:ts": "turbo run test --output-logs=errors-only",
+ "test:rs": "cargo nextest run --workspace --status-level=fail --final-status-level=fail",
+ "test:affected": "turbo run test --affected --output-logs=errors-only",
"check": "bun run --parallel check:ts check:rs",
"check:ts": "bun run --parallel check:tools check:ws",
"check:tools": "biome check . --no-errors-on-unmatched && bun run check:types",
@@ -38,9 +39,9 @@
"fmt:rs": "cargo fmt --all",
"fix": "bun run --parallel fix:ts fix:rs",
"fix:ts": "bun run --parallel fix:tools fix:ws",
- "fix:tools": "biome check --write --unsafe . --no-errors-on-unmatched",
+ "fix:tools": "bun run fmt:tools && biome check --write --unsafe . --no-errors-on-unmatched",
"fix:ws": "turbo run fix",
- "fix:rs": "cargo fmt --all && cargo clippy --fix --allow-dirty --all-targets --no-deps --allow-staged --broken-code --allow-no-vcs",
+ "fix:rs": "cargo fmt --all && cargo clippy --workspace --fix --allow-dirty --all-targets --no-deps --allow-staged --broken-code --allow-no-vcs",
"ci:check:affected": "bun run check:affected",
"ci:check:full": "bun run check:ts",
"ci:build:native": "bun scripts/ci-build-native.ts",
@@ -67,7 +68,8 @@
"@biomejs/biome": "^2.4",
"@bufbuild/protoc-gen-es": "^2.11",
"@types/bun": "^1.3",
- "@typescript/native-preview": "^7.0.0-dev.20260302.1",
+ "@typescript/native-preview": "7.0.0-dev.20260322.1",
+ "typescript": "^6.0.2",
"lint-staged": "^16.3",
"prettier": "^3.8",
"turbo": "^2.5.8"
diff --git a/packages/coding-agent/src/prompts/tools/chunk-edit.md b/packages/coding-agent/src/prompts/tools/chunk-edit.md
index a3fe843a3..5ba6c421b 100644
--- a/packages/coding-agent/src/prompts/tools/chunk-edit.md
+++ b/packages/coding-agent/src/prompts/tools/chunk-edit.md
@@ -48,8 +48,12 @@ You **MUST** use the narrowest region that covers your change. Replacing without
- `@decl` — everything except leading trivia (signature + body + closing delimiter).
- *(no region)* — the entire chunk including leading trivia. Same as `@head` + `@body` + `@tail`.
+**`@decl` excludes leading trivia** (doc comments, decorators, attributes). In Rust, `#[attr]` is trivia; in Python, `@decorator` is trivia; in TypeScript, JSDoc + decorators are trivia. To modify attributes or decorators, target `@head` instead. Including them in `@decl` replacement content **will duplicate them** because the original trivia is preserved.
+
For leaf chunks (fields, variants, single-line items), `@body` falls back to the full chunk.
+In TypeScript, if a doc comment or decorator is separated from the function by a blank line, it may appear as its own orphan `chunk` node instead of being absorbed into the function's `@head`. Use the `?` selector (`read(path="file", sel="?")`) to discover the actual chunk layout before editing.
+
`append`/`prepend` without a `@region` inserts _outside_ the chunk. To add children _inside_ a class, struct, enum, or function body, use `@body`:
- `class_Foo@body` + `append` → adds inside the class before `}`
- `class_Foo@body` + `prepend` → adds inside the class after `{`
diff --git a/scripts/rate-edit-tool.py b/scripts/rate-edit-tool.py
index d4c2874bc..6d233c7b2 100755
--- a/scripts/rate-edit-tool.py
+++ b/scripts/rate-edit-tool.py
@@ -57,6 +57,8 @@ MODELS = [
"openrouter/minimax/minimax-m2.7",
]
+ORACLE_MODEL = "openrouter/anthropic/claude-opus-4.6"
+
PROMPT = textwrap.dedent(
"""\
You are evaluating the current code-reading and code-editing tools on the files in this directory.
@@ -111,6 +113,49 @@ FINAL_REVIEW_PROMPT = textwrap.dedent(
"""
).strip()
+ORACLE_REVIEW_PROMPT = textwrap.dedent(
+ """\
+
+ You are the oracle reviewer for a tool-evaluation benchmark.
+ You are reading multiple independent reviews of the same edit-tool session.
+ Synthesize only what is supported by the supplied reviews.
+
+ This matters. Be specific and conservative.
+
+
+
+ 1. Deduplicate overlapping observations into one finding.
+ 2. Separate well-supported findings from weaker signals. When evidence is mixed or thin, lower confidence instead of overstating.
+ 3. Cite which review models reported each finding and summarize the concrete evidence they observed.
+ 4. End with practical improvement areas prioritized by expected trust and usability gains.
+
+ Output markdown only with this exact structure:
+
+ # Oracle synthesis
+
+ ## Findings
+ - Finding: ...
+ - Confidence: High | Medium | Low
+ - Evidence: cite the models and the behavior they observed
+ - Improvement area: the concrete product or UX change this finding points to
+
+ ## Improvement areas
+ 1. ...
+ 2. ...
+
+ If the reviews do not support any concrete finding, say so explicitly under `## Findings`.
+
+
+
+ {{REVIEWS}}
+
+
+
+ Use only the supplied reviews. Output markdown only.
+
+ """
+).strip()
+
TODOS = [
"Map the current read and edit tool surface area on main.ts, main.rs, main.py, and main.md.",
"Exercise supported read and edit paths with concrete before/after verification across code and prose fixtures.",
@@ -1445,6 +1490,68 @@ def run_model_sync(
session_state=session_state,
)
+def build_oracle_review_prompt(results: list[ModelResult]) -> str:
+ review_sections: list[str] = []
+ for result in sorted(results, key=lambda candidate: candidate.model):
+ review_text = Path(result.review_path).read_text(encoding="utf-8").strip()
+ if not review_text:
+ continue
+ review_sections.append(
+ textwrap.dedent(
+ f"""\
+
+ {result.model}
+ {result.review_path}
+
+ {review_text}
+
+ """
+ ).strip()
+ )
+
+ if not review_sections:
+ raise ValueError("No review content available for oracle synthesis")
+
+ review_payload = "\n\n".join(review_sections)
+ return ORACLE_REVIEW_PROMPT.replace("{{REVIEWS}}", review_payload)
+
+
+
+def run_oracle_review_sync(
+ *,
+ model: str,
+ omp_bin: str,
+ results: list[ModelResult],
+ results_dir: Path,
+ timeout: float,
+ openrouter_key: str,
+) -> Path:
+ oracle_path = results_dir / f"review_oracle_{slugify(shorten_model_name(model))}.md"
+ prompt = build_oracle_review_prompt(results)
+
+ with RpcClient(
+ executable=omp_bin,
+ model=model,
+ cwd=results_dir,
+ env={"OPENROUTER_API_KEY": openrouter_key, "PI_STRICT_EDIT_MODE": "1"},
+ thinking="high",
+ tools=(),
+ no_skills=True,
+ no_rules=True,
+ no_session=True,
+ startup_timeout=30.0,
+ request_timeout=30.0,
+ ) as client:
+ client.install_headless_ui()
+ client.prompt_and_wait(prompt, timeout=timeout)
+ review_markdown = client.get_last_assistant_text()
+
+ if not isinstance(review_markdown, str) or not review_markdown.strip():
+ raise RpcError("Oracle model completed without synthesis text")
+
+ oracle_path.write_text(review_markdown.strip() + "\n", encoding="utf-8")
+ return oracle_path
+
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Run OpenRouter fixture evaluations through omp RPC mode.")
parser.add_argument("--omp-bin", default=os.environ.get("OMP_BIN"))
@@ -1452,6 +1559,7 @@ def parse_args() -> argparse.Namespace:
parser.add_argument("--results-dir")
parser.add_argument("--timeout", type=float, default=900.0, help="Per-model timeout in seconds.")
parser.add_argument("--model", dest="models", action="append", help="Repeat to limit execution to specific models.")
+ parser.add_argument("--oracle-model", default=ORACLE_MODEL, help="Model used to synthesize findings across all reviews.")
return parser.parse_args()
@@ -1491,7 +1599,19 @@ async def run_all(args: argparse.Namespace) -> int:
if failures:
printer.finish(f"{failures}/{len(results)} model run(s) failed")
return 1
- printer.finish(f"{len(results)} review file(s) written to {results_dir}")
+
+ oracle_path = await asyncio.to_thread(
+ run_oracle_review_sync,
+ model=args.oracle_model,
+ omp_bin=omp_bin,
+ results=results,
+ results_dir=results_dir,
+ timeout=args.timeout,
+ openrouter_key=openrouter_key,
+ )
+ printer.finish(
+ f"{len(results)} review file(s) and oracle synthesis written to {results_dir} ({oracle_path.name})"
+ )
return 0